{"id":"2ome-lm-2025","kind":"source","name":"2OMe-LM: predicting 2′-O-methylation sites in human RNA using a pre-trained RNA language model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12342186/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf417","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"54fe4db6f35c03d0d4f3ef4da720eb26a832199372c56d0956609ff07af750ee","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12342186/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.332Z","legacy_paper":{"id":"2ome-lm-2025","title":"2OMe-LM: predicting 2′-O-methylation sites in human RNA using a pre-trained RNA language model","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12342186/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf417","notes":"Numeric result checked against Table 1. in primary full-text XML; journal/source: Bioinformatics."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"adar-gpt-editing-2026","kind":"source","name":"ADAR-GPT: A continually fine-tuned language model for predicting A-to-I RNA editing sites","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12798952/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1073/pnas.2529073123","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"cc8c7eb928f246f1f347a8822f614cd3475381c35eef6d579032ce441580198e","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12798952/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"adar-gpt-editing-2026","title":"ADAR-GPT: A continually fine-tuned language model for predicting A-to-I RNA editing sites","year":2026,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12798952/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Proceedings of the National Academy of Sciences of the United States of America; PMC ID: PMC12798952. RNA editing site benchmark on a restricted liver validation set.","doi":"10.1073/pnas.2529073123"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"akscore-2020","kind":"source","name":"AK-Score: Accurate Protein-Ligand Binding Affinity Prediction Using an Ensemble of 3D-Convolutional Neural Networks","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7697539/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.3390/ijms21228424","publication_status":"peer_reviewed","year":2020,"artifact_sha256":"40cfd28dcd587599768ec99a6590ec593486475ff01c7b1d1f229b44aa91bf8d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7697539/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.436853+00:00","legacy_paper":{"id":"akscore-2020","title":"AK-Score: Accurate Protein-Ligand Binding Affinity Prediction Using an Ensemble of 3D-Convolutional Neural Networks","year":2020,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7697539/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: International Journal of Molecular Sciences; PMC ID: PMC7697539.","doi":"10.3390/ijms21228424"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"antibody-deamidation-plm-2024","kind":"source","name":"The Accurate Prediction of Antibody Deamidations by Combining High-Throughput Automated Peptide Mapping and Protein Language Model-Based Deep Learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11417914/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.3390/antib13030074","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"aa049f6d78e29540ba902a0d3b7f53d49e9833e4dad9869f79ac27664e8c150b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11417914/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.478Z","legacy_paper":{"id":"antibody-deamidation-plm-2024","title":"The Accurate Prediction of Antibody Deamidations by Combining High-Throughput Automated Peptide Mapping and Protein Language Model-Based Deep Learning","year":2024,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11417914/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.3390/antib13030074","notes":"Numeric result checked against Table 1 in primary full-text XML; journal/source: Antibodies."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"antibody-flexibility-2025","kind":"source","name":"Enhancing antibody-antigen interaction prediction with atomic flexibility","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12530544/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1371/journal.pcbi.1013576","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"57e64694c69052ed0495570e12ebfb4bb6c0ad152219f23827cd4b1cb53450ef","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12530544/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.400Z","legacy_paper":{"id":"antibody-flexibility-2025","title":"Enhancing antibody-antigen interaction prediction with atomic flexibility","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12530544/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: PLOS Computational Biology; PMC ID: PMC12530544.","doi":"10.1371/journal.pcbi.1013576"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"arsenal-regulatory-dna-2026","kind":"source","name":"Short-Context Regulatory DNA Language Models with Motif-Discovery Regularization","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12889687/","version":"preprint version in PMC","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.64898/2026.02.05.703637","publication_status":"preprint","year":2026,"artifact_sha256":"4a264956e47fc633aaff6573aac368dc691dd5de709b27c7421c078608ff542a","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12889687/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"arsenal-regulatory-dna-2026","title":"Short-Context Regulatory DNA Language Models with Motif-Discovery Regularization","year":2026,"publication_status":"preprint","version":"preprint version in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12889687/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: bioRxiv; PMC ID: PMC12889687. Preprint; result is a supervised downstream model rather than a general-purpose DNA foundation model.","doi":"10.64898/2026.02.05.703637"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"b2-2ome-lm-2025","kind":"result","name":"2OMe-LM · AUC · human RNA 2OMe sites","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["human RNA 2-prime-O-methylation site prediction"]},"source_ids":["2ome-lm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-2ome-lm-2025"}],"attributes":{"printed_value":"0.919","numeric_value":"0.919","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, 2OMe-LM row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.332Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 1, 2OMe-LM row, AUC column; cell: 0.919","artifact_sha256":"54fe4db6f35c03d0d4f3ef4da720eb26a832199372c56d0956609ff07af750ee","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12342186/fullTextXML"},"legacy_id":"b2-2ome-lm-2025","legacy_row":{"id":"b2-2ome-lm-2025","paper_id":"2ome-lm-2025","domain_id":"rna-transcriptomes","task":"human RNA 2-prime-O-methylation site prediction","model":"2OMe-LM","model_version":"not stated in table","dataset":"human RNA 2OMe sites","dataset_version":"","split":"5-fold cross-validation","metric":"AUC","value":"0.919","unit":"fraction","uncertainty":"","protocol":"pretrained RNA language model predictor","source_locator":"Table 1, 2OMe-LM row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12342186/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-antibody-deamidation-plm-2024","kind":"result","name":"ESM-2 650M embeddings + classifier · accuracy · antibody peptide-mapping training dataset","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["antibody deamidation-site prediction"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-antibody-deamidation-plm-2024"}],"attributes":{"printed_value":"0.944","numeric_value":"0.944","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":"± 0.012","source_locator":"Table 1, Global embeddings only row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.478Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 1, Global embeddings only row, Accuracy column; cell: 0.944 ± 0.012","artifact_sha256":"aa049f6d78e29540ba902a0d3b7f53d49e9833e4dad9869f79ac27664e8c150b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11417914/fullTextXML"},"legacy_id":"b2-antibody-deamidation-plm-2024","legacy_row":{"id":"b2-antibody-deamidation-plm-2024","paper_id":"antibody-deamidation-plm-2024","domain_id":"proteins-complexes","task":"antibody deamidation-site prediction","model":"ESM-2 650M embeddings + classifier","model_version":"esm2_t33_650m_UR50D","dataset":"antibody peptide-mapping training dataset","dataset_version":"","split":"fivefold stratified CV","metric":"accuracy","value":"0.944","unit":"fraction","uncertainty":"± 0.012","protocol":"global contextual embeddings only","source_locator":"Table 1, Global embeddings only row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11417914/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"b2-barcodebert-2026","kind":"result","name":"BarcodeBERT (4–4-4) · accuracy · DNA barcodes of unseen species","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["unseen-species genus classification"]},"source_ids":["barcodebert-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-barcodebert-2026"}],"attributes":{"printed_value":"78.5","numeric_value":"78.5","metric":"accuracy","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 1, BarcodeBERT (4–4-4) row, unseen-species genus-level 1-NN Acc (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558051+00:00","notes":"Resolved the two-level column header: Acc (%) falls under genus-level 1-NN probe of unseen species, not seen-species classification or BIN reconstruction. BarcodeBERT (4–4-4) has 78.5 in this cell.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 78.5.","artifact_sha256":"493f9fe70b483780ba76d51ccf217d3ca83539c82b89917fd3ccebe2b6eb831d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13008329/fullTextXML"},"legacy_id":"b2-barcodebert-2026","legacy_row":{"id":"b2-barcodebert-2026","paper_id":"barcodebert-2026","domain_id":"dna-genomes","task":"unseen-species genus classification","model":"BarcodeBERT (4–4-4)","model_version":"4–4–4","dataset":"DNA barcodes of unseen species","dataset_version":"","split":"1-NN probe","metric":"accuracy","value":"78.5","unit":"percent","uncertainty":"","protocol":"genus-level nearest-neighbor probe on species unseen in training","source_locator":"Table 1, BarcodeBERT (4–4-4) row, unseen-species genus-level 1-NN Acc (%) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13008329/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-birna-bert-2025","kind":"result","name":"BiRNA-BERT · F1 · extremely long-sequence species classification","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["extremely long RNA species classification"]},"source_ids":["birna-bert-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-birna-bert-2025"}],"attributes":{"printed_value":"0.804","numeric_value":"0.804","metric":"F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, BiRNA-BERT row, F1 Score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.292Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, BiRNA-BERT row, F1 Score column; cell: 0.804","artifact_sha256":"bf7dbc52b6515301c77010c513f13e676c38395ddc82c20310171f518690c152","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12635123/fullTextXML"},"legacy_id":"b2-birna-bert-2025","legacy_row":{"id":"b2-birna-bert-2025","paper_id":"birna-bert-2025","domain_id":"rna-transcriptomes","task":"extremely long RNA species classification","model":"BiRNA-BERT","model_version":"not stated in table","dataset":"extremely long-sequence species classification","dataset_version":"","split":"paper evaluation","metric":"F1","value":"0.804","unit":"fraction","uncertainty":"","protocol":"adaptive tokenization on full-length long RNA sequences","source_locator":"Table 2, BiRNA-BERT row, F1 Score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12635123/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-cathe2-2025","kind":"result","name":"CATHe2 + ProstT5 · F1 · CATH superfamily benchmark","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["CATH superfamily annotation"]},"source_ids":["cathe2-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-cathe2-2025"}],"attributes":{"printed_value":"82.3","numeric_value":"82.3","metric":"F1","metric_direction":"unknown","unit":"percent","uncertainty":"± 1.3 percentage points","source_locator":"Table 3, ProstT5 full row, F1 score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.366Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, ProstT5 full row, F1 score column; cell: 82.3% ± 1.3%","artifact_sha256":"713dbfb6ec1cc1aa85c0543eb93aafa0b45b8873df28053b765dd0a1b6d9b563","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12631783/fullTextXML"},"legacy_id":"b2-cathe2-2025","legacy_row":{"id":"b2-cathe2-2025","paper_id":"cathe2-2025","domain_id":"proteins-complexes","task":"CATH superfamily annotation","model":"CATHe2 + ProstT5","model_version":"full ProstT5","dataset":"CATH superfamily benchmark","dataset_version":"","split":"paper evaluation","metric":"F1","value":"82.3","unit":"percent","uncertainty":"± 1.3 percentage points","protocol":"amino-acid and structural alphabet embedding classifier","source_locator":"Table 3, ProstT5 full row, F1 score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12631783/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"b2-clathrin-plm-2025","kind":"result","name":"ESM-2 embedding + paper classifier · accuracy · CLA-IND0.6","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["clathrin protein classification"]},"source_ids":["clathrin-plm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-clathrin-plm-2025"}],"attributes":{"printed_value":"0.916","numeric_value":"0.916","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, Independent test / ESM-2 row, ACC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558194+00:00","notes":"Resolved the blank evaluation-strategy cells by their independent-test row group. ESM-2 ACC is 0.916 there; the cross-validation ESM-2 ACC is instead 0.873. This is the paper classifier using embeddings, not a standalone checkpoint.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.916.","artifact_sha256":"2edc86b25707c1b737d26117093ce8d856e79cc5d0b335f27c1c341f887f1c7e","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12238356/fullTextXML"},"legacy_id":"b2-clathrin-plm-2025","legacy_row":{"id":"b2-clathrin-plm-2025","paper_id":"clathrin-plm-2025","domain_id":"proteins-complexes","task":"clathrin protein classification","model":"ESM-2 embedding + paper classifier","model_version":"not stated in table","dataset":"CLA-IND0.6","dataset_version":"","split":"independent test","metric":"accuracy","value":"0.916","unit":"fraction","uncertainty":"","protocol":"single-feature ESM-2 embedding comparison","source_locator":"Table 2, Independent test / ESM-2 row, ACC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12238356/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-cobra-rna-binding-2026","kind":"result","name":"ERNIE-RNA + CoBRA · MCC · CoBRA compound-binding test set","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA compound-binding site prediction"]},"source_ids":["cobra-rna-binding-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-cobra-rna-binding-2026"}],"attributes":{"printed_value":"0.657","numeric_value":"0.657","metric":"MCC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558197+00:00","notes":"Matched ERNIE-RNA jointly with TCL focal loss, then the MCC column. Table 2 explicitly reports test-set models. The cell is 0.657, distinct from AUROC 0.868.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.657.","artifact_sha256":"8c6a6f00f5fa5f62acf301a66e9e6fa9ef11c7a05ad9b7447d2ade2ce8eba793","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12790621/fullTextXML"},"legacy_id":"b2-cobra-rna-binding-2026","legacy_row":{"id":"b2-cobra-rna-binding-2026","paper_id":"cobra-rna-binding-2026","domain_id":"rna-transcriptomes","task":"RNA compound-binding site prediction","model":"ERNIE-RNA + CoBRA","model_version":"not stated in table","dataset":"CoBRA compound-binding test set","dataset_version":"","split":"test set","metric":"MCC","value":"0.657","unit":"unitless","uncertainty":"","protocol":"ERNIE-RNA embedding with TCL focal loss","source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12790621/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-codonbert-vaccines-2024","kind":"result","name":"CodonBERT · Spearman rho · flu-vaccine sequences","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["flu-vaccine mRNA property prediction"]},"source_ids":["codonbert-vaccines-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-codonbert-vaccines-2024"}],"attributes":{"printed_value":"0.81","numeric_value":"0.81","metric":"Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558201+00:00","notes":"Matched the CodonBERT row and Flu vaccines column (0.81). The table footnote identifies regression columns as Spearman rank correlation and singles out E. coli as classification; this is not a flu-vaccine accuracy score.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.81.","artifact_sha256":"2968073753e6d44feff9c08b131edf23145e95b171434539dddf77bb92847033","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11368176/fullTextXML"},"legacy_id":"b2-codonbert-vaccines-2024","legacy_row":{"id":"b2-codonbert-vaccines-2024","paper_id":"codonbert-vaccines-2024","domain_id":"rna-transcriptomes","task":"flu-vaccine mRNA property prediction","model":"CodonBERT","model_version":"not stated in table","dataset":"flu-vaccine sequences","dataset_version":"","split":"paper evaluation","metric":"Spearman rho","value":"0.81","unit":"unitless","uncertainty":"","protocol":"codon-based model fine-tuned for downstream regression","source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11368176/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-dart-eval-regulatory-2024","kind":"result","name":"DNABERT-2 · accuracy · DART-Eval cCREs versus matched shuffled controls","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory element identification"]},"source_ids":["dart-eval-regulatory-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-dart-eval-regulatory-2024"}],"attributes":{"printed_value":"0.876","numeric_value":"0.876","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558203+00:00","notes":"Inspected the pinned NeurIPS primary PDF table and explanatory text. The DNABERT-2 row reports 0.876 under Zero-Shot Accuracy. The caption defines this as pairwise prioritization of positives over matched controls, distinct from supervised absolute accuracy.","evidence":"DNABERT-2; zero-shot accuracy 0.876; probed absolute/paired 0.847/0.943; fine-tuned absolute/paired 0.913/0.973.","artifact_sha256":"e5aee5b1f7cc6fd961b1d2a131d02cf243b79e091d5e418fbabee7fde9b39b22","retrieval_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf"},"legacy_id":"b2-dart-eval-regulatory-2024","legacy_row":{"id":"b2-dart-eval-regulatory-2024","paper_id":"dart-eval-regulatory-2024","domain_id":"dna-genomes","task":"regulatory element identification","model":"DNABERT-2","model_version":"not stated in table","dataset":"DART-Eval cCREs versus matched shuffled controls","dataset_version":"","split":"paper evaluation","metric":"accuracy","value":"0.876","unit":"fraction","uncertainty":"","protocol":"zero-shot likelihood ranking: higher likelihood for cCRE than matched control","source_locator":"Table 3, DNABERT-2 row, Zero-Shot Accuracy column","source_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-dnabert2-enhancer-2025","kind":"result","name":"DNABERT2-Enhancer · AUC · Liu training dataset","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer recognition"]},"source_ids":["dnabert2-enhancer-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-dnabert2-enhancer-2025"}],"attributes":{"printed_value":"0.965","numeric_value":"0.965","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558204+00:00","notes":"Resolved the first-layer row group. DNABERT2-Enhancer AUC is 0.965, whereas second-layer AUC is 0.933. The caption explicitly describes 5-fold cross-validation on Liu training data, not an independent held-out test.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.965.","artifact_sha256":"d052b80efe7bfc1380994ad28503a5575f04ef940f74d5c9c137cb4ba6827863","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11981215/fullTextXML"},"legacy_id":"b2-dnabert2-enhancer-2025","legacy_row":{"id":"b2-dnabert2-enhancer-2025","paper_id":"dnabert2-enhancer-2025","domain_id":"dna-genomes","task":"enhancer recognition","model":"DNABERT2-Enhancer","model_version":"not stated in table","dataset":"Liu training dataset","dataset_version":"","split":"5-fold cross-validation","metric":"AUC","value":"0.965","unit":"fraction","uncertainty":"","protocol":"first-layer enhancer versus non-enhancer classifier","source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11981215/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-eden-genomic-classification-2026","kind":"result","name":"DNABERT-2 · MCC · GUE H-CPD","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["human core-promoter classification"]},"source_ids":["eden-genomic-classification-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-eden-genomic-classification-2026"}],"attributes":{"printed_value":"70.52","numeric_value":"70.52","metric":"MCC","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:37.531Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, DNABERT-2 row, H-CPD (MCC) column; cell: 70.52","artifact_sha256":"38a6e26b3caffe8e021a2b0b672218e783aca9ee42046765e323946813015e65","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12879454/fullTextXML"},"legacy_id":"b2-eden-genomic-classification-2026","legacy_row":{"id":"b2-eden-genomic-classification-2026","paper_id":"eden-genomic-classification-2026","domain_id":"dna-genomes","task":"human core-promoter classification","model":"DNABERT-2","model_version":"not stated in table","dataset":"GUE H-CPD","dataset_version":"","split":"paper evaluation","metric":"MCC","value":"70.52","unit":"percent","uncertainty":"","protocol":"DNABERT-2 comparator in consolidated H-CPD table; rerun provenance not explicit","source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12879454/","evaluation_origin":"paper_compilation","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-ernie-rna-2025","kind":"result","name":"ERNIE-RNA · binary F1 · bpRNA-new","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary-structure prediction"]},"source_ids":["ernie-rna-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-ernie-rna-2025"}],"attributes":{"printed_value":"0.575","numeric_value":"0.575","metric":"binary F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558206+00:00","notes":"Resolved bpRNA-new as the first three-column dataset group and F1-Score (binary) as its third metric. ERNIE-RNA zero shot is 86M and reports 0.575; RNA3DB-2D F1 is instead 0.542.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.575.","artifact_sha256":"0bd1d4b3cbf5d59d452cec4864614947861efcee050ba07e7de395cd90630047","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12627772/fullTextXML"},"legacy_id":"b2-ernie-rna-2025","legacy_row":{"id":"b2-ernie-rna-2025","paper_id":"ernie-rna-2025","domain_id":"rna-transcriptomes","task":"RNA secondary-structure prediction","model":"ERNIE-RNA","model_version":"86M","dataset":"bpRNA-new","dataset_version":"","split":"cross-family test","metric":"binary F1","value":"0.575","unit":"fraction","uncertainty":"","protocol":"zero-shot attention-derived base-pair prediction","source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12627772/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-esm2-ofs-fitness-2025","kind":"result","name":"ESM2 OFS pseudo-perplexity · Spearman rho · ProteinGym substitutions","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant fitness prediction"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-esm2-ofs-fitness-2025"}],"attributes":{"printed_value":"0.403","numeric_value":"0.403","metric":"Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Publisher PDF retrieved through official APS harvest endpoint after direct download returned403. Table I is ProteinGym substitutions, not indels TableII. Last column aggregate mean0.403; separate function categories precede it. This verifies reported score, not experimental reproduction. Comparator rows in this table are sourced from ProteinGym; OFS PP is authors own method.","evidence":"Headers: Activity(43), Binding(14), Expression(17), Organismal fitness(77), Stability(66), Aggregate mean. ESM2:OFS PP row:0.393,0.279,0.397,0.331,0.507,0.403. Verified publisher PDF layout extraction against web-rendered primary PDF table.","artifact_sha256":"085ef646f11b8e5335c4b3d86b15fb6c7bf5edf4a80a8b622753ac69d9991a67","retrieval_url":"https://harvest.aps.org/v2/journals/articles/10.1103/zhx7-hcmm/fulltext"},"legacy_id":"b2-esm2-ofs-fitness-2025","legacy_row":{"id":"b2-esm2-ofs-fitness-2025","paper_id":"esm2-ofs-fitness-2025","domain_id":"proteins-complexes","task":"protein variant fitness prediction","model":"ESM2 OFS pseudo-perplexity","model_version":"not stated in table","dataset":"ProteinGym substitutions","dataset_version":"","split":"aggregate across assays","metric":"Spearman rho","value":"0.403","unit":"unitless","uncertainty":"","protocol":"authors’ zero-shot ESM2 OFS pseudo-perplexity evaluation; aggregate mean across ProteinGym substitution assays","source_locator":"Table I, ESM2: OFS PP row, Aggregate Mean Spearman correlation column","source_url":"https://journals.aps.org/prxlife/pdf/10.1103/zhx7-hcmm","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-fusion-breakpoint-foundation-models-2026","kind":"result","name":"Nucleotide Transformer + NN (middle) · ROC AUC · gene fusion breakpoint DNA sequences","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["gene fusion breakpoint classification"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-fusion-breakpoint-foundation-models-2026"}],"attributes":{"printed_value":"0.994","numeric_value":"0.994","metric":"ROC AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558209+00:00","notes":"Matched NT jointly with NN (middle) and ROC AUC 0.994 in the full-test-set table. NT with SVM reports 0.995 and is a separate pipeline.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.994.","artifact_sha256":"0f4d9de77f1e39cfd2164a20653d86370767da684dc22d17e09f589761abeb5f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13182013/fullTextXML"},"legacy_id":"b2-fusion-breakpoint-foundation-models-2026","legacy_row":{"id":"b2-fusion-breakpoint-foundation-models-2026","paper_id":"fusion-breakpoint-foundation-models-2026","domain_id":"dna-genomes","task":"gene fusion breakpoint classification","model":"Nucleotide Transformer + NN (middle)","model_version":"not stated in table","dataset":"gene fusion breakpoint DNA sequences","dataset_version":"","split":"full test set","metric":"ROC AUC","value":"0.994","unit":"fraction","uncertainty":"","protocol":"middle embedding with neural-network classifier","source_locator":"Table 2, NT / NN (middle) row, ROC AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13182013/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-genomic-tokenizer-selection-2025","kind":"result","name":"Caduceus (character tokens) · MCC · genomic benchmark categories","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory sequence classification"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-genomic-tokenizer-selection-2025"}],"attributes":{"printed_value":"0.778","numeric_value":"0.778","metric":"MCC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558210+00:00","notes":"Matched Regulatory row with Caduceus (char) column, 0.778. Caption establishes these as MCC summaries by category; model-size row identifies 3.9M parameters. This is an aggregated category result, not a single unspecified split.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.778.","artifact_sha256":"0a01c36fdd63f3f6db509777e61c3f87e8a298c810f8aef7974915aaa0655342","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12453675/fullTextXML"},"legacy_id":"b2-genomic-tokenizer-selection-2025","legacy_row":{"id":"b2-genomic-tokenizer-selection-2025","paper_id":"genomic-tokenizer-selection-2025","domain_id":"dna-genomes","task":"regulatory sequence classification","model":"Caduceus (character tokens)","model_version":"3.9M parameter variant","dataset":"genomic benchmark categories","dataset_version":"","split":"paper benchmark summary","metric":"MCC","value":"0.778","unit":"unitless","uncertainty":"","protocol":"task-category MCC across benchmark datasets","source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12453675/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-gsmformer-ppi-2026","kind":"result","name":"GSMFormer-PPI + ProstT5 · AUROC · paper PPI test set","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein interaction prediction"]},"source_ids":["gsmformer-ppi-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-gsmformer-ppi-2026"}],"attributes":{"printed_value":"0.988","numeric_value":"0.988","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 6, ProstT5 embedding row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558212+00:00","notes":"Matched ProstT5 embedding row and AUROC column, 0.988. Caption explicitly describes GSMFormer-PPI using embeddings as node features, not standalone ProstT5 prediction.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.988.","artifact_sha256":"9b364b5d73d16f2787f93f78f17dbe98b954ab9c2c64c1df960eec2e615eb3b4","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12873117/fullTextXML"},"legacy_id":"b2-gsmformer-ppi-2026","legacy_row":{"id":"b2-gsmformer-ppi-2026","paper_id":"gsmformer-ppi-2026","domain_id":"proteins-complexes","task":"protein-protein interaction prediction","model":"GSMFormer-PPI + ProstT5","model_version":"not stated in table","dataset":"paper PPI test set","dataset_version":"","split":"test set","metric":"AUROC","value":"0.988","unit":"fraction","uncertainty":"","protocol":"ProstT5 embeddings as graph node features","source_locator":"Table 6, ProstT5 embedding row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12873117/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-megsite-2025","kind":"result","name":"MegSite + ESM3 · AUC · DNA-129_Test","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["DNA-binding residue prediction"]},"source_ids":["megsite-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-megsite-2025"}],"attributes":{"printed_value":"0.948","numeric_value":"0.948","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558213+00:00","notes":"Resolved DNA-129_Test row group and ESM3 row. AUC is 0.948; the next numeric cell 0.582 is AP. Caption states an embedding comparison within MegSite.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.948.","artifact_sha256":"10d13122331813243d83b84fe6f9294eac7e7c03cde082ebed276191ac41089c","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12496013/fullTextXML"},"legacy_id":"b2-megsite-2025","legacy_row":{"id":"b2-megsite-2025","paper_id":"megsite-2025","domain_id":"proteins-complexes","task":"DNA-binding residue prediction","model":"MegSite + ESM3","model_version":"not stated in table","dataset":"DNA-129_Test","dataset_version":"","split":"independent test","metric":"AUC","value":"0.948","unit":"fraction","uncertainty":"","protocol":"ESM3 multimodal embedding ablation in MegSite","source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12496013/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-mrna-lm-2025","kind":"result","name":"mRNA-LM · Spearman rho · mRNA half-life","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA half-life prediction"]},"source_ids":["mrna-lm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-mrna-lm-2025"}],"attributes":{"printed_value":"0.696","numeric_value":"0.696","metric":"Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558214+00:00","notes":"Resolved mRNA half-life column under the Spearman header spanning three tasks. mRNA-LM gives 0.696. Caption identifies average test performance across cross-validation splits; protein-expression AUROC is a different column.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.696.","artifact_sha256":"3a23de3c672ec162d13561c483f180a73b550d717256deffdc9099accec205fd","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11962594/fullTextXML"},"legacy_id":"b2-mrna-lm-2025","legacy_row":{"id":"b2-mrna-lm-2025","paper_id":"mrna-lm-2025","domain_id":"rna-transcriptomes","task":"mRNA half-life prediction","model":"mRNA-LM","model_version":"not stated in table","dataset":"mRNA half-life","dataset_version":"","split":"test set across CV splits","metric":"Spearman rho","value":"0.696","unit":"unitless","uncertainty":"","protocol":"average test performance across cross-validation splits","source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11962594/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-mrnabert-2025","kind":"result","name":"mRNABERT · R-squared · human ultra-long mRNAs","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["translation-efficiency prediction"]},"source_ids":["mrnabert-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-mrnabert-2025"}],"attributes":{"printed_value":"0.669","numeric_value":"0.669","metric":"R-squared","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558216+00:00","notes":"Resolved Human group and its R-squared subcolumn. mRNABERT (3066) reports 0.669; Human Spearman is 0.814 and Mouse R-squared is 0.649. Caption specifies ultra-long mRNA translation-efficiency prediction.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.669.","artifact_sha256":"ff08ba895b7080446c08a930548b48a0041ae990c222ebb07e6ba7dcaf48ad44","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12644827/fullTextXML"},"legacy_id":"b2-mrnabert-2025","legacy_row":{"id":"b2-mrnabert-2025","paper_id":"mrnabert-2025","domain_id":"rna-transcriptomes","task":"translation-efficiency prediction","model":"mRNABERT","model_version":"3066-nt input","dataset":"human ultra-long mRNAs","dataset_version":"","split":"paper evaluation","metric":"R-squared","value":"0.669","unit":"unitless","uncertainty":"","protocol":"human translation-efficiency regression at 3066-nt input","source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12644827/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-mulan-2025","kind":"result","name":"MULAN-ESM2 S · AUC · HumanPPI","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["human protein-protein interaction prediction"]},"source_ids":["mulan-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-mulan-2025"}],"attributes":{"printed_value":"0.717","numeric_value":"0.717","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558217+00:00","notes":"Resolved the multirow header: HumanPPI uses AUC. MULAN-ESM2 S has 0.717; this is the small-model group, distinct from M and L variants.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.717.","artifact_sha256":"771a9a26ebda6f49ea266540e8dd6e6de0cbaef724de818ca6124a5f9c50d350","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12452268/fullTextXML"},"legacy_id":"b2-mulan-2025","legacy_row":{"id":"b2-mulan-2025","paper_id":"mulan-2025","domain_id":"proteins-complexes","task":"human protein-protein interaction prediction","model":"MULAN-ESM2 S","model_version":"small ESM2 backbone","dataset":"HumanPPI","dataset_version":"","split":"paper evaluation","metric":"AUC","value":"0.717","unit":"fraction","uncertainty":"","protocol":"MULAN sequence-structure model based on ESM2 8M","source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12452268/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-phylogpn-2025","kind":"result","name":"PhyloGPN · AUROC · ClinVar 3-prime UTR variants","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["ClinVar 3-prime UTR variant classification"]},"source_ids":["phylogpn-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-phylogpn-2025"}],"attributes":{"printed_value":"0.94","numeric_value":"0.94","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558218+00:00","notes":"Matched 3-prime UTR row and PhyloGPN column (0.94). Caption specifies log-likelihood-ratio predictions of ClinVar classes and explicitly defines each cell as AUROC.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.94.","artifact_sha256":"807f3a26cbfa9b5ce238d92164bd523302c67d1c5794b08273c51cca1acd4224","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11908359/fullTextXML"},"legacy_id":"b2-phylogpn-2025","legacy_row":{"id":"b2-phylogpn-2025","paper_id":"phylogpn-2025","domain_id":"dna-genomes","task":"ClinVar 3-prime UTR variant classification","model":"PhyloGPN","model_version":"not stated in table","dataset":"ClinVar 3-prime UTR variants","dataset_version":"","split":"paper evaluation","metric":"AUROC","value":"0.94","unit":"fraction","uncertainty":"","protocol":"log-likelihood-ratio scoring","source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11908359/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-polya-glm-2025","kind":"result","name":"HyenaDNA · AUC · poly(A) Gene-Gene","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["polyadenylation site detection"]},"source_ids":["polya-glm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-polya-glm-2025"}],"attributes":{"printed_value":"0.7510","numeric_value":"0.7510","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558220+00:00","notes":"Resolved Few-shot group, HyenaDNA row, and G-G subcolumn under AUC (0.7510). IG-G AUC is 0.7541. Caption states averages over five-fold cross-validation and distinguishes negative sampling regions.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.7510.","artifact_sha256":"e9ebd53d88837ad8d457881ffee918d2734dcae87d3c5cd03135947b6cf5dbde","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12799945/fullTextXML"},"legacy_id":"b2-polya-glm-2025","legacy_row":{"id":"b2-polya-glm-2025","paper_id":"polya-glm-2025","domain_id":"dna-genomes","task":"polyadenylation site detection","model":"HyenaDNA","model_version":"not stated in table","dataset":"poly(A) Gene-Gene","dataset_version":"","split":"5-fold cross-validation","metric":"AUC","value":"0.7510","unit":"fraction","uncertainty":"","protocol":"few-shot Gene-Gene negative-set comparison","source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12799945/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-rlsite-rna-binding-2025","kind":"result","name":"RLsite · AUC · T18","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA-small-molecule binding-site prediction"]},"source_ids":["rlsite-rna-binding-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-rlsite-rna-binding-2025"}],"attributes":{"printed_value":"0.828","numeric_value":"0.828","metric":"AUC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, RLsite row, T18 AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558222+00:00","notes":"Matched RLsite and AUC (0.828). Caption explicitly identifies dataset T18; MCC 0.474 is a different metric.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.828.","artifact_sha256":"a50f344e253162ae43f51d7120cfb35a1d0f6114fd8176d760aceb6d05fd95bd","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12417085/fullTextXML"},"legacy_id":"b2-rlsite-rna-binding-2025","legacy_row":{"id":"b2-rlsite-rna-binding-2025","paper_id":"rlsite-rna-binding-2025","domain_id":"rna-transcriptomes","task":"RNA-small-molecule binding-site prediction","model":"RLsite","model_version":"not stated in table","dataset":"T18","dataset_version":"","split":"paper evaluation","metric":"AUC","value":"0.828","unit":"fraction","uncertainty":"","protocol":"RNA language-model plus graph-attention classifier","source_locator":"Table 1, RLsite row, T18 AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12417085/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-rnaret-2026","kind":"result","name":"RNAret · F1 · MirTarRAW","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["miRNA-mRNA interaction prediction"]},"source_ids":["rnaret-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-rnaret-2026"}],"attributes":{"printed_value":"0.9622","numeric_value":"0.9622","metric":"F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558224+00:00","notes":"Resolved the MirTarRAW section, 5-mer RNAret row, and F1 column (0.9622), distinct from DeepMirTarLeft F1 0.9728. Methods confirm 72/8/20 train/validation/test partition for MirTarRAW.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.9622.","artifact_sha256":"e970e7322e07fb3c9d12efd315691cc5de5575a3f2616f4b788614c8c706dd0b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13111708/fullTextXML"},"legacy_id":"b2-rnaret-2026","legacy_row":{"id":"b2-rnaret-2026","paper_id":"rnaret-2026","domain_id":"rna-transcriptomes","task":"miRNA-mRNA interaction prediction","model":"RNAret","model_version":"5-mer","dataset":"MirTarRAW","dataset_version":"","split":"held-out test","metric":"F1","value":"0.9622","unit":"fraction","uncertainty":"","protocol":"5-mer RNAret classifier; 72/8/20 train/validation/test split","source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13111708/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-spin-protein-function-2026","kind":"result","name":"SPIN + ESM2-35M · F1 macro-weighted · TRX","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein function annotation"]},"source_ids":["spin-protein-function-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-spin-protein-function-2026"}],"attributes":{"printed_value":"0.796","numeric_value":"0.796","metric":"F1 macro-weighted","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558225+00:00","notes":"Resolved Test group and macro-weighted F1 subcolumn (0.796) for frozen ESM2-35M in SPIN. Test weighted accuracy is 0.798. Methods define inverse-frequency class weighting for macro-weighted F1.","evidence":"Original PMC XML table headers, row groups and caption inspected; printed value 0.796.","artifact_sha256":"9701843e93bf7fa3ead71e19693fb07d483f1022379871adfb04486783722a9d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12970593/fullTextXML"},"legacy_id":"b2-spin-protein-function-2026","legacy_row":{"id":"b2-spin-protein-function-2026","paper_id":"spin-protein-function-2026","domain_id":"proteins-complexes","task":"protein function annotation","model":"SPIN + ESM2-35M","model_version":"ESM2-35M frozen","dataset":"TRX","dataset_version":"","split":"test set","metric":"F1 macro-weighted","value":"0.796","unit":"fraction","uncertainty":"","protocol":"frozen ESM2-35M backbone in SPIN","source_locator":"Table 1, ESM2-35M Test row, F1_m-w column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12970593/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"b2-structure-informed-plm-2025","kind":"result","name":"structure-informed pLM · AUROC · variant-effects benchmark","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant-effect classification"]},"source_ids":["structure-informed-plm-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-b2-structure-informed-plm-2025"}],"attributes":{"printed_value":".803","numeric_value":"0.803","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Full-text HTML succeeds although EuropePMC XMLreturned404. Row is mutation-site variables AA+SS+RSA+CM, not neighbouring environment variant. AUROC .803 is numerically equivalent to preserved legacy0.803. Source check, not experimental reproduction; do not claim original source printed leading zero.","evidence":"Table4 headers: Type, Variable(s), Spearman rho, AUROC, AUPRC. Parsed HTML row: AA+SS+RSA+CM | .552 | .803 | .792. Primary web rendering independently confirms columns.","artifact_sha256":"76082e1cd992d2c09c38f86d05aba575cc76c5022b53a297123b713bb1ce9267","retrieval_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/"},"legacy_id":"b2-structure-informed-plm-2025","legacy_row":{"id":"b2-structure-informed-plm-2025","paper_id":"structure-informed-plm-2025","domain_id":"proteins-complexes","task":"protein variant-effect classification","model":"structure-informed pLM","model_version":"not stated in table","dataset":"variant-effects benchmark","dataset_version":"","split":"paper evaluation","metric":"AUROC","value":"0.803","unit":"fraction","uncertainty":"","protocol":"combined amino-acid, secondary structure, solvent accessibility and contact-map scoring","source_locator":"Table 4, AA+SS+RSA+CM row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:33:26Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"barcodebert-2026","kind":"source","name":"BarcodeBERT: transformers for biodiversity analyses","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13008329/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioadv/vbag054","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"493f9fe70b483780ba76d51ccf217d3ca83539c82b89917fd3ccebe2b6eb831d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13008329/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558051+00:00","legacy_paper":{"id":"barcodebert-2026","title":"BarcodeBERT: transformers for biodiversity analyses","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13008329/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioadv/vbag054","notes":"Numeric result checked against Table 1 in primary full-text XML; journal/source: Bioinformatics Advances."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"birna-bert-2025","kind":"source","name":"BiRNA-BERT allows efficient RNA language modeling with adaptive tokenization","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12635123/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s42003-025-08982-0","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"bf7dbc52b6515301c77010c513f13e676c38395ddc82c20310171f518690c152","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12635123/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.292Z","legacy_paper":{"id":"birna-bert-2025","title":"BiRNA-BERT allows efficient RNA language modeling with adaptive tokenization","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12635123/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s42003-025-08982-0","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Communications Biology."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"boltz-stereochemistry-2025","kind":"source","name":"Improving Stereochemical Limitations in Protein–Ligand Complex Structure Prediction","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12658688/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acsomega.5c07675","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"78a77b9a0ab8bfa371f5b9baef3f443f4590d6e71cf864d67e90e9ebdfa7fc1b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12658688/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.555674+00:00","legacy_paper":{"id":"boltz-stereochemistry-2025","title":"Improving Stereochemical Limitations in Protein–Ligand Complex Structure Prediction","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12658688/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: ACS Omega; PMC ID: PMC12658688.","doi":"10.1021/acsomega.5c07675"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"boltz1-2025","kind":"source","name":"Boltz-1 Democratizing Biomolecular Interaction Modeling","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11601547/","version":"PMC archival version PMC11601547.4","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2024.11.19.624167","publication_status":"preprint","year":2025,"artifact_sha256":"1ebf712314d9a1c678ded989cc95a0c00c0331e5ad8c9f63194bc9780971d214","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11601547/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.424864+00:00","legacy_paper":{"id":"boltz1-2025","title":"Boltz-1 Democratizing Biomolecular Interaction Modeling","year":2025,"publication_status":"preprint","version":"PMC archival version PMC11601547.4","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11601547/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11601547.","doi":"10.1101/2024.11.19.624167"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"bpfold-2025","kind":"source","name":"Deep generalizable prediction of RNA secondary structure via base pair motif energy","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12216785/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1038/s41467-025-60048-1","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"976218bd172998a1a6e7ed1609ecb8cb2ee380fb48a8dc7b25bc05ea8b0a49af","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12216785/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.502000+00:00","legacy_paper":{"id":"bpfold-2025","title":"Deep generalizable prediction of RNA secondary structure via base pair motif energy","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12216785/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Nature Communications; PMC ID: PMC12216785.","doi":"10.1038/s41467-025-60048-1"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cammiq-2022","kind":"source","name":"Strain level microbial detection and quantification with applications to single cell metagenomics","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9616933/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1038/s41467-022-33869-7","publication_status":"peer_reviewed","year":2022,"artifact_sha256":"f0939647ed3de995d58254f79472a612c21b0e1b2560a82783302aa1a148dde3","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9616933/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.408237+00:00","legacy_paper":{"id":"cammiq-2022","title":"Strain level microbial detection and quantification with applications to single cell metagenomics","year":2022,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9616933/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Nature Communications; PMC ID: PMC9616933.","doi":"10.1038/s41467-022-33869-7"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"catalog-baseline-kraken2","kind":"baseline","name":"Kraken2","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-kraken2"],"links":[{"relation":"model","target_id":"catalog-model-kraken2"},{"relation":"applicable_to","target_id":"catalog-task-heldout-clade"},{"relation":"applicable_to","target_id":"catalog-task-phage-pathogen-reads"}],"attributes":{"baseline_type":"established_method","applicability":"proposed","requirements":"Public classifier; database build/version must be pinned separately.","missing_metadata":{"exact_protocol":"not_yet_extracted"}}} {"id":"catalog-baseline-scvi","kind":"baseline","name":"scVI","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-scvi"],"links":[{"relation":"model","target_id":"catalog-model-scvi"},{"relation":"applicable_to","target_id":"catalog-task-cell-reference-mapping"},{"relation":"applicable_to","target_id":"catalog-task-cell-batch-integration"}],"attributes":{"baseline_type":"established_method","applicability":"proposed","requirements":"Public software; train a task-specific model on the permitted split.","missing_metadata":{"exact_protocol":"not_yet_extracted"}}} {"id":"catalog-baseline-vina","kind":"baseline","name":"AutoDock Vina","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-vina"],"links":[{"relation":"model","target_id":"catalog-model-vina"},{"relation":"applicable_to","target_id":"catalog-task-ligand-pose"}],"attributes":{"baseline_type":"established_method","applicability":"proposed","requirements":"Public docking software; receptor and ligand preparation required.","missing_metadata":{"exact_protocol":"not_yet_extracted"}}} {"id":"catalog-model-alphafold-3-server","kind":"model","name":"AlphaFold 3 Server","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["foundation model"]},"source_ids":["catalog-source-alphafold-3-server"],"links":[],"attributes":{"entity_level":"family","version":"hosted server","reported_name":"AlphaFold 3 Server","access":"Manual, non-commercial server access; output terms restrict automated docking combinations.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"AlphaFold 3 Server is a candidate method in the molecular-interactions catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to AlphaFold 3 Server official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-alphafold-3-server"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"hosted server","source_ids":["catalog-source-alphafold-3-server"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-alphafold-3-server"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'hosted server' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-alphagenome","kind":"model","name":"AlphaGenome","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-alphagenome"],"links":[],"attributes":{"entity_level":"family","version":"API / released weights","reported_name":"AlphaGenome","access":"Rate-limited, non-commercial API requires a key. Downloadable weights require accepting non-commercial model terms; local inference recommends an H100 GPU.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"AlphaGenome is a candidate method in the dna-genomes catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to AlphaGenome official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-alphagenome"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"API / released weights","source_ids":["catalog-source-alphagenome"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-alphagenome"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'API / released weights' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-boltz-2","kind":"model","name":"Boltz-2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["foundation model"]},"source_ids":["catalog-source-boltz-2"],"links":[{"relation":"variant_of","target_id":"discovery-model-boltz"}],"attributes":{"entity_level":"family","version":"released weights","reported_name":"Boltz-2","access":"Public MIT code and weights; substantial compute required.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Boltz is a biomolecular interaction model family. Boltz-2 adds affinity prediction to complex-structure prediction.","sections":[{"title":"Boltz-2 architecture","body":"The Boltz-2 implementation combines molecular and alignment features with a Pairformer module. A conditioned diffusion module predicts coordinates; a separate affinity module produces binding outputs. This architecture description applies to Boltz-2, not automatically to every Boltz family release.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"src/boltz/model/models/boltz2.py at the pinned repository revision: MSAModule, PairformerModule, DiffusionConditioning and AffinityModule; README Inference"},{"title":"How it works","body":"The documented YAML input describes the biomolecules and requested properties. Structure prediction and affinity outputs are distinct: one affinity output estimates binding strength, while another classifies binders against decoys.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"facts":[{"label":"Access","value":"Repository states code and models use the MIT licence","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"},{"label":"Configuration in this record","value":"released weights","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"strengths":[{"text":"Supports structure and affinity workflows in one openly distributed project.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"limitations":[{"text":"Binder probability and affinity regression are trained with different supervision and must not be compared as the same metric. Unqualified CLI calls select the latest model.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"diagram":{"title":"Conceptual procedure","steps":["Molecular inputs","MSA / Pairformer features","Coordinate diffusion","Structure","Separate affinity module"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"Pinned repository src/boltz/model/models/boltz2.py, module construction and forward; README affinity prediction"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-chai-1","kind":"model","name":"Chai-1","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes","molecular-interactions"],"method_types":["foundation model"]},"source_ids":["catalog-source-chai-1"],"links":[],"attributes":{"entity_level":"family","version":"released weights","reported_name":"Chai-1","access":"Public code and weights under Apache 2.0; substantial compute required.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Chai-1 is a candidate method in the proteins-complexes, molecular-interactions catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to Chai-1 official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-chai-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"released weights","source_ids":["catalog-source-chai-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-chai-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'released weights' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-diffdock-l","kind":"model","name":"DiffDock-L","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["specialist"]},"source_ids":["catalog-source-diffdock-l"],"links":[],"attributes":{"entity_level":"family","version":"2024 release","reported_name":"DiffDock-L","access":"Public pose-prediction code and weights; no native affinity prediction.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"DiffDock-L is a candidate method in the molecular-interactions catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to DiffDock-L official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-diffdock-l"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"2024 release","source_ids":["catalog-source-diffdock-l"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-diffdock-l"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration '2024 release' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-dnabert-2","kind":"model","name":"DNABERT-2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-dnabert-2"],"links":[{"relation":"variant_of","target_id":"discovery-model-dnabert-2"}],"attributes":{"entity_level":"family","version":"117M","reported_name":"DNABERT-2","access":"Public checkpoint; remote model code needs review before local use.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"DNABERT-2 is a DNA encoder pretrained on sequences from multiple species. It supplies representations that can be adapted to genomic tasks.","sections":[{"title":"How it works","body":"DNA is compressed into variable-length byte-pair tokens. A BERT-style encoder uses ALiBi positional biases; masked-language pretraining learns contextual features. Sequence pooling or a separately trained prediction head produces task outputs.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"facts":[{"label":"Released model","value":"DNABERT-2-117M","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"},{"label":"Objective","value":"Masked-language pretraining","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"},{"label":"Configuration in this record","value":"117M","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"strengths":[{"text":"One released encoder can support embedding extraction and supervised adaptation across several genomic tasks.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"limitations":[{"text":"An encoder embedding is not a splice-impact prediction. Pooling, sequence context and the supervised head are part of the evaluated pipeline.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"diagram":{"title":"Conceptual procedure","steps":["DNA sequence","Byte-pair tokens","BERT encoder with ALiBi","Token or pooled embeddings","Task-specific head"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-esm-2","kind":"model","name":"ESM-2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes"],"method_types":["foundation model"]},"source_ids":["catalog-source-esm-2"],"links":[{"relation":"variant_of","target_id":"discovery-model-esm-2"}],"attributes":{"entity_level":"family","version":"8M","reported_name":"ESM-2","access":"Public checkpoint; small 8M variant suits a local pilot.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"ESM-2 is a family of protein sequence transformers that produce residue-level and sequence-level representations.","sections":[{"title":"How it works","body":"Amino-acid tokens pass through a pretrained transformer. Hidden states can be retained for each residue or pooled for a whole protein; downstream tasks need an explicit scoring rule or predictor. ESMFold adds a structure-prediction system and is a separate pipeline.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"facts":[{"label":"Training resource","value":"UniRef-derived protein sequences","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"},{"label":"Configuration distinction","value":"The catalogue 8M entry is not the 650M or 15B checkpoint","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"},{"label":"Configuration in this record","value":"8M","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"strengths":[{"text":"Embeddings can be extracted directly from individual sequences; the repository provides several model sizes.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"limitations":[{"text":"Different parameter sizes, pooling methods and supervised heads are not interchangeable evaluations.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"diagram":{"title":"Conceptual procedure","steps":["Protein sequence","Amino-acid tokens","ESM-2 transformer","Residue embeddings","Pooling or task predictor"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-esmfold","kind":"model","name":"ESMFold","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes"],"method_types":["foundation model"]},"source_ids":["catalog-source-esmfold"],"links":[{"relation":"variant_of","target_id":"discovery-model-esmfold"}],"attributes":{"entity_level":"family","version":"v1","reported_name":"ESMFold","access":"Public checkpoint; materially larger than ESM-2 8M.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"ESMFold predicts protein structures from individual amino-acid sequences using an ESM-2 representation model and a folding system.","sections":[{"title":"How it works","body":"The sequence is embedded and converted into a three-dimensional structure. The implementation exposes recycling and chunking controls; those choices affect memory use and the exact evaluated run.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"facts":[{"label":"Primary output","value":"PDB structure with confidence information","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"},{"label":"Configuration in this record","value":"v1","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"strengths":[{"text":"The documented inference interface produces a structure directly from sequence.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"limitations":[{"text":"Long sequences and larger batches can exceed device memory. Version v0 and v1 refer to different released models.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"diagram":{"title":"Conceptual procedure","steps":["Protein sequence","ESM-2 features","Folding system","Recycling","Predicted structure"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-evo-2","kind":"model","name":"Evo 2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes","microbes-communities"],"method_types":["foundation model"]},"source_ids":["catalog-source-evo-2"],"links":[],"attributes":{"entity_level":"family","version":"7B","reported_name":"Evo 2","access":"Public checkpoints; official local inference needs CUDA hardware and substantial memory.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Evo 2 models and generates DNA at nucleotide resolution using the StripedHyena 2 architecture.","sections":[{"title":"How it works","body":"An autoregressive sequence model predicts successive nucleotides from preceding context. Scoring and generation use the selected released checkpoint; context length and device requirements depend on that configuration.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"facts":[{"label":"Training resource","value":"OpenGenome2","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"},{"label":"Objective","value":"Autoregressive sequence prediction","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"},{"label":"Configuration in this record","value":"7B","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"strengths":[{"text":"The family is designed for long-context sequence modelling, with released inference code.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"limitations":[{"text":"A family-level maximum context length does not establish the settings used by a particular published evaluation.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"diagram":{"title":"Conceptual procedure","steps":["DNA nucleotides","StripedHyena 2","Autoregressive predictions","Sequence scoring or generation"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-gears","kind":"model","name":"GEARS","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["specialist"]},"source_ids":["catalog-source-gears"],"links":[{"relation":"family","target_id":"discovery-model-gears"}],"attributes":{"entity_level":"family","version":"published implementation","reported_name":"GEARS","access":"Public code; task-specific training data required.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"GEARS predicts transcriptional responses to genetic perturbations using single-cell perturbation-screen data.","sections":[{"title":"How it works","body":"A task-specific model is trained on measured perturbations, then predicts gene-expression responses for requested single or combined perturbations. Training composition determines what generalisation question is being tested.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"facts":[{"label":"Required evidence","value":"Perturbation identities and cells per condition","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"},{"label":"Configuration in this record","value":"published implementation","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"strengths":[{"text":"The implementation explicitly supports single-gene and multi-gene perturbation workflows.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"limitations":[{"text":"The maintainers state that cross-cell-type transfer is unsupported and that reliable combinatorial prediction needs some combinatorial training data.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"diagram":{"title":"Conceptual procedure","steps":["Perturbation-screen cells","Training perturbations","GEARS predictor","Requested perturbation","Expression response"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-geneformer","kind":"model","name":"Geneformer","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["foundation model"]},"source_ids":["catalog-source-geneformer"],"links":[],"attributes":{"entity_level":"family","version":"published checkpoints","reported_name":"Geneformer","access":"Public checkpoints; specify exact version before evaluation.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Geneformer is a candidate method in the cells-tissues catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to Geneformer official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-geneformer"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"published checkpoints","source_ids":["catalog-source-geneformer"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-geneformer"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'published checkpoints' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-kraken2","kind":"model","name":"Kraken2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["baseline"]},"source_ids":["catalog-source-kraken2"],"links":[],"attributes":{"entity_level":"family","version":"current database pinned at run time","reported_name":"Kraken2","access":"Public classifier; database build/version must be pinned separately.","method_type":"baseline","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Kraken2 is a candidate method in the microbes-communities catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to Kraken2 official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-kraken2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"current database pinned at run time","source_ids":["catalog-source-kraken2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-kraken2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'current database pinned at run time' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-metagene-1","kind":"model","name":"METAGENE-1","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["foundation model"]},"source_ids":["catalog-source-metagene-1"],"links":[],"attributes":{"entity_level":"family","version":"6B","reported_name":"METAGENE-1","access":"Public Apache 2.0 checkpoint; 512-token context and large local memory requirement.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"METAGENE-1 is a candidate method in the microbes-communities catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to METAGENE-1 official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-metagene-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"6B","source_ids":["catalog-source-metagene-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-metagene-1"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration '6B' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-metaphlan","kind":"model","name":"MetaPhlAn","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["specialist"]},"source_ids":["catalog-source-metaphlan"],"links":[],"attributes":{"entity_level":"family","version":"current marker database pinned at run time","reported_name":"MetaPhlAn","access":"Public profiler; marker database version must be pinned separately.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"MetaPhlAn is a candidate method in the microbes-communities catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to MetaPhlAn official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-metaphlan"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"current marker database pinned at run time","source_ids":["catalog-source-metaphlan"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-metaphlan"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'current marker database pinned at run time' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-mimic","kind":"model","name":"MIMIC","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes","proteins-complexes"],"method_types":["foundation model"]},"source_ids":["catalog-source-mimic"],"links":[],"attributes":{"entity_level":"family","version":"1.0","reported_name":"MIMIC","access":"Public MIT code and 1.25B-parameter weights; large local memory requirement.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"MIMIC is a candidate method in the rna-transcriptomes, proteins-complexes catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to MIMIC official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-mimic"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"1.0","source_ids":["catalog-source-mimic"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-mimic"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration '1.0' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-mrna-fm","kind":"model","name":"mRNA-FM","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-mrna-fm"],"links":[{"relation":"variant_of","target_id":"discovery-model-rna-fm"}],"attributes":{"entity_level":"family","version":"codon-tokenised","reported_name":"mRNA-FM","access":"Public checkpoint trained on coding sequences (CDS); input must be codon aligned. UTR-only sequences are outside its training modality.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"mRNA-FM is the coding-sequence extension of RNA-FM, intended to represent messenger RNA coding regions.","sections":[{"title":"How it works","body":"Coding sequences are converted into model tokens and processed by the pretrained sequence encoder. Its output supplies embeddings for a downstream predictor. The input preparation differs from the ncRNA model.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"}],"facts":[{"label":"Training modality","value":"Repository reports 45 million mRNA coding sequences","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"},{"label":"Configuration in this record","value":"codon-tokenised","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"}],"strengths":[{"text":"Provides a representation specifically pretrained on coding RNA.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"}],"limitations":[{"text":"Coding-sequence training does not establish performance on UTR-only inputs or other non-coding RNA.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"}],"diagram":{"title":"Conceptual procedure","steps":["Coding sequence","mRNA-FM input tokens","Pretrained encoder","Embeddings","Downstream predictor"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction, mRNA-FM subsection"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-nt-v2","kind":"model","name":"Nucleotide Transformer v2","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-nt-v2"],"links":[],"attributes":{"entity_level":"family","version":"50M multi-species","reported_name":"Nucleotide Transformer v2","access":"Public checkpoint.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Nucleotide Transformer v2 is a candidate method in the dna-genomes catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to Nucleotide Transformer v2 official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-nt-v2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"50M multi-species","source_ids":["catalog-source-nt-v2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-nt-v2"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration '50M multi-species' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-pangolin","kind":"model","name":"Pangolin","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["specialist"]},"source_ids":["catalog-source-pangolin"],"links":[{"relation":"family","target_id":"discovery-model-pangolin"}],"attributes":{"entity_level":"family","version":"published checkpoints","reported_name":"Pangolin","access":"Public specialist code and models under GPL-3.0.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"Pangolin predicts changes in splice-site strength from DNA variants. It accepts variant files or custom sequence inputs.","sections":[{"title":"Architecture","body":"Pangolin uses 16 residual blocks with dilated convolutions and skip connections. Separate outputs estimate splice-site probability and usage across heart, liver, brain and testis. The published model was trained using sequence and splicing measurements from human, rhesus macaque, rat and mouse.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"Original paper linked in README: https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture; Results: Pangolin predicts splice site usage"},{"title":"How it works","body":"Reference genome and transcript annotation define the sequence context. The neural predictor estimates splice-site strength; the command-line tool reports the largest positive and negative changes near each variant. Masking optionally removes particular gains and losses at annotated sites.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"facts":[{"label":"Masking","value":"Default mask=True; a mask=False evaluation is a distinct configuration","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"},{"label":"Configuration in this record","value":"published checkpoints","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"strengths":[{"text":"Provides changes in splice strength and their positions, with configurable scoring distance and masking.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"limitations":[{"text":"Only substitutions and simple indels are supported by the documented interface. Missing gene annotations, reference mismatches and chromosome-edge cases can exclude variants.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"diagram":{"title":"Conceptual procedure","steps":["DNA context","Dilated residual convolutions","Tissue-specific outputs","Splice strength","Variant-induced change"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"Original paper linked in README: https://doi.org/10.1186/s13059-022-02664-4, Figure 1 and Methods: Deep neural network architecture; README Usage"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-prokbert","kind":"model","name":"ProkBERT","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["microbes-communities"],"method_types":["foundation model"]},"source_ids":["catalog-source-prokbert"],"links":[],"attributes":{"entity_level":"family","version":"mini","reported_name":"ProkBERT","access":"Public model family and mini checkpoint.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"ProkBERT is a candidate method in the microbes-communities catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to ProkBERT official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-prokbert"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"mini","source_ids":["catalog-source-prokbert"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-prokbert"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'mini' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-proteinmpnn","kind":"model","name":"ProteinMPNN","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["proteins-complexes"],"method_types":["specialist"]},"source_ids":["catalog-source-proteinmpnn"],"links":[{"relation":"variant_of","target_id":"discovery-model-proteinmpnn"}],"attributes":{"entity_level":"family","version":"v_48_020","reported_name":"ProteinMPNN","access":"Public code and checkpoints; requires a suitable protein structure.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"ProteinMPNN designs amino-acid sequences for a supplied protein backbone, with controls for fixed residues and chains.","sections":[{"title":"How it works","body":"A parsed structure and design constraints are supplied to the sequence-design model. It samples amino-acid sequences conditional on the backbone; sampling temperature changes diversity. Full-backbone and Cα-only weights are separate configurations.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"facts":[{"label":"Catalogue weight name","value":"v_48_020","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"},{"label":"Output","value":"Designed sequences and model scores","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"},{"label":"Configuration in this record","value":"v_48_020","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"strengths":[{"text":"Allows selected chains and positions to be redesigned while retaining specified residues.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"limitations":[{"text":"Requires a suitable input structure. Sequence generation does not itself demonstrate folding, activity or experimental success.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"diagram":{"title":"Conceptual procedure","steps":["Backbone structure","Chain / residue constraints","ProteinMPNN","Conditional sequence sampling","Designed sequences"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-rhofold","kind":"model","name":"RhoFold+","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes"],"method_types":["specialist"]},"source_ids":["catalog-source-rhofold"],"links":[],"attributes":{"entity_level":"family","version":"pretrained","reported_name":"RhoFold+","access":"Public code and checkpoint instructions.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"RhoFold+ is a candidate method in the rna-transcriptomes catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to RhoFold+ official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-rhofold"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"pretrained","source_ids":["catalog-source-rhofold"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-rhofold"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'pretrained' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-rna-fm","kind":"model","name":"RNA-FM","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["rna-transcriptomes"],"method_types":["foundation model"]},"source_ids":["catalog-source-rna-fm"],"links":[{"relation":"family","target_id":"discovery-model-rna-fm"}],"attributes":{"entity_level":"family","version":"ncRNA","reported_name":"RNA-FM","access":"Public code and checkpoint instructions.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"RNA-FM is a pretrained RNA sequence encoder for structural and functional representation learning.","sections":[{"title":"How it works","body":"A BERT-style transformer encodes RNA tokens into contextual embeddings after self-supervised sequence training. Structural or functional predictions require the corresponding downstream model; RNA-FM alone should not be labelled as a complete 3D folding pipeline.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"facts":[{"label":"RNA-FM training","value":"Repository reports more than 23 million non-coding RNA sequences","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"},{"label":"Configuration in this record","value":"ncRNA","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"strengths":[{"text":"Reusable representations do not require experimental labels during pretraining.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"limitations":[{"text":"The ncRNA encoder and the coding-sequence mRNA-FM extension have different training modalities and should not share checkpoint identities.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"diagram":{"title":"Conceptual procedure","steps":["RNA sequence","RNA tokens","Pretrained transformer","Contextual embeddings","Task-specific predictor"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-scfoundation","kind":"model","name":"scFoundation","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["foundation model"]},"source_ids":["catalog-source-scfoundation"],"links":[],"attributes":{"entity_level":"family","version":"100M","reported_name":"scFoundation","access":"Public code; model weights have separate terms that must be checked.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"scFoundation is a candidate method in the cells-tissues catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to scFoundation official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-scfoundation"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"100M","source_ids":["catalog-source-scfoundation"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-scfoundation"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration '100M' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-scgpt","kind":"model","name":"scGPT","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["foundation model"]},"source_ids":["catalog-source-scgpt"],"links":[{"relation":"variant_of","target_id":"discovery-model-scgpt"}],"attributes":{"entity_level":"family","version":"whole-human","reported_name":"scGPT","access":"Public code and downloadable checkpoints; use the unfine-tuned whole-human model for a new task.","method_type":"foundation model","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"scGPT learns representations of genes and cells from single-cell measurements. Its pretrained checkpoints support task-specific adaptation.","sections":[{"title":"How it works","body":"Gene identifiers and expression values are encoded together and processed by a transformer. The resulting representations support cell embeddings or task heads. Vocabulary, preprocessing and the selected checkpoint must accompany any result.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"facts":[{"label":"Whole-human training","value":"Repository reports 33 million normal human cells","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"},{"label":"Configuration in this record","value":"whole-human","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"strengths":[{"text":"The repository provides whole-human and specialised checkpoints, plus workflows for annotation, integration and perturbation tasks.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"limitations":[{"text":"Whole-human, organ-specific and continually pretrained checkpoints are different configurations. A pretraining claim does not establish transfer performance in a new cell population.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"diagram":{"title":"Conceptual procedure","steps":["Gene IDs and expression","Gene / value encoders","Transformer","Cell and gene representations","Adapted task output"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-scvi","kind":"model","name":"scVI","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["cells-tissues"],"method_types":["baseline"]},"source_ids":["catalog-source-scvi"],"links":[],"attributes":{"entity_level":"family","version":"scvi-tools","reported_name":"scVI","access":"Public software; train a task-specific model on the permitted split.","method_type":"baseline","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"scVI is a candidate method in the cells-tissues catalogue. The linked resource identifies the project; a checkpoint-level profile has not yet been extracted.","sections":[{"title":"Available evidence","body":"The discovery record links to scVI official resource. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["catalog-source-scvi"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"facts":[{"label":"Recorded configuration","value":"scvi-tools","source_ids":["catalog-source-scvi"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["catalog-source-scvi"],"source_locator":"Official resource linked by the discovery record; model-specific architecture not extracted"}],"coverage":"limited","gaps":["The linked official resource has not yielded a reviewed architecture or procedure for the configuration 'scvi-tools' in this release.","Unresolved registry fields: checkpoint revision (not yet extracted), training data (not yet extracted), licence (not yet extracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"catalog-model-spliceai","kind":"model","name":"SpliceAI","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["dna-genomes"],"method_types":["specialist"]},"source_ids":["catalog-source-spliceai"],"links":[{"relation":"variant_of","target_id":"discovery-model-spliceai"}],"attributes":{"entity_level":"family","version":"1.3.1","reported_name":"SpliceAI","access":"Public archived code under PolyForm Strict; model weights are CC BY-NC 4.0 for non-commercial use.","method_type":"specialist","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"SpliceAI predicts how sequence variants alter splice-site usage. Its variant annotation tool combines reference sequence with gene annotation.","sections":[{"title":"Architecture","body":"SpliceAI uses a residual convolutional network with dilated filters to integrate sequence context. It predicts donor, acceptor and non-splice-site probabilities along the sequence; comparing alleles converts those predictions into variant scores. The output is not tissue-specific.","source_ids":["src-discovery-illumina-spliceai","src-discovery-tkzeng-pangolin"],"source_locator":"SpliceAI README and linked Jaganathan et al. paper; Pangolin primary paper https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture, direct comparison with SpliceAI"},{"title":"How it works","body":"The tool evaluates reference and alternative alleles and reports predicted acceptor/donor gains and losses with their relative positions. Annotation, search distance and masking determine the reported variant scores.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"facts":[{"label":"Inputs","value":"VCF, reference FASTA and gene annotation","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"},{"label":"Outputs","value":"Acceptor/donor gain and loss delta scores","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"},{"label":"Configuration in this record","value":"1.3.1","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"strengths":[{"text":"Produces splice-specific scores and predicted event positions without fitting a classifier to the user’s assay labels.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"limitations":[{"text":"Code, trained models and precomputed annotations have distinct use terms. Annotation and sequence checks can leave variants unscored.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"diagram":{"title":"Conceptual procedure","steps":["Reference / alternate DNA","Dilated residual convolutions","Acceptor / donor probabilities","Allelic difference","Variant delta scores"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-illumina-spliceai","src-discovery-tkzeng-pangolin"],"source_locator":"SpliceAI README and linked Jaganathan et al. paper; Pangolin primary paper https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture, direct comparison with SpliceAI"},"coverage":"reviewed","gaps":["An immutable hash for the exact 1.3.1 weight files has not been attached to this family record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-model-vina","kind":"model","name":"AutoDock Vina","description":"Existing candidate catalogue entry; exact checkpoint and metadata require primary-source review.","status":"discovered","facets":{"areas":["molecular-interactions"],"method_types":["baseline"]},"source_ids":["catalog-source-vina"],"links":[],"attributes":{"entity_level":"family","version":"1.2.7","reported_name":"AutoDock Vina","access":"Public docking software; receptor and ligand preparation required.","method_type":"baseline","missing_metadata":{"checkpoint_revision":"not_yet_extracted","training_data":"not_yet_extracted","licence":"not_yet_extracted"},"profile":{"summary":"AutoDock Vina is a conventional docking engine for searching ligand conformations against a receptor.","sections":[{"title":"How it works","body":"A scoring function guides gradient-based conformational search. Candidate poses are ranked under the selected docking configuration; receptor preparation and search settings are part of the method.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"facts":[{"label":"Method class","value":"Docking search with a scoring function","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"},{"label":"Configuration in this record","value":"1.2.7","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"strengths":[{"text":"Provides a procedural comparator with batch and multiple-ligand workflows.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"limitations":[{"text":"A docking score is not an experimentally measured affinity. Pose and affinity tasks require separate evaluation.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"diagram":{"title":"Conceptual procedure","steps":["Prepared receptor and ligand","Search configuration","Conformation search","Scoring function","Ranked docking poses"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"catalog-source-alphafold-3-server","kind":"source","name":"AlphaFold 3 Server official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://alphafoldserver.com/output-terms","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-alphagenome","kind":"source","name":"AlphaGenome official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/google-deepmind/alphagenome_research","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-boltz-2","kind":"source","name":"Boltz-2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/jwohlwend/boltz","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-chai-1","kind":"source","name":"Chai-1 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/chaidiscovery/chai-lab","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-diffdock-l","kind":"source","name":"DiffDock-L official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/gcorso/DiffDock","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-dnabert-2","kind":"source","name":"DNABERT-2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/zhihan1996/DNABERT-2-117M","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-esm-2","kind":"source","name":"ESM-2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/facebookresearch/esm","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-esmfold","kind":"source","name":"ESMFold official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/facebookresearch/esm","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-evo-2","kind":"source","name":"Evo 2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ArcInstitute/evo2","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-gears","kind":"source","name":"GEARS official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/snap-stanford/GEARS","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-geneformer","kind":"source","name":"Geneformer official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/ctheodoris/Geneformer","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-kraken2","kind":"source","name":"Kraken2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/DerrickWood/kraken2","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-metagene-1","kind":"source","name":"METAGENE-1 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/metagene-ai/METAGENE-1","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-metaphlan","kind":"source","name":"MetaPhlAn official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biobakery/MetaPhlAn","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-mimic","kind":"source","name":"MIMIC official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/polymathic-ai/MIMIC","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-mrna-fm","kind":"source","name":"mRNA-FM official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RNA-FM","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-nt-v2","kind":"source","name":"Nucleotide Transformer v2 official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://huggingface.co/InstaDeepAI/nucleotide-transformer-v2-50m-multi-species","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-pangolin","kind":"source","name":"Pangolin official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/tkzeng/Pangolin","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-prokbert","kind":"source","name":"ProkBERT official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/nbrg-ppcu/prokbert","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-proteinmpnn","kind":"source","name":"ProteinMPNN official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/dauparas/ProteinMPNN","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-rhofold","kind":"source","name":"RhoFold+ official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RhoFold","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-rna-fm","kind":"source","name":"RNA-FM official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ml4bio/RNA-FM","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-scfoundation","kind":"source","name":"scFoundation official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/biomap-research/scFoundation","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-scgpt","kind":"source","name":"scGPT official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/bowang-lab/scGPT","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-scvi","kind":"source","name":"scVI official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/scverse/scvi-tools","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-spliceai","kind":"source","name":"SpliceAI official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/Illumina/SpliceAI","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-source-vina","kind":"source","name":"AutoDock Vina official resource","description":"","status":"discovered","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/ccsb-scripps/AutoDock-Vina","version":null,"missing_metadata":{"version":"not_pinned_in_legacy_catalogue"},"retrieved_at":"2026-09-16T10:50:02Z"}} {"id":"catalog-task-cell-batch-integration","kind":"benchmark","name":"Batch integration","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-scgpt","catalog-source-scvi"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Batch integration","scope_note":"Test whether cell identity is retained across donors and batches.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Batch integration asks whether data from different experiments can be combined without losing biological differences.","sections":[{"title":"Procedure","body":"Evaluate both sides of the problem: cells should no longer separate mainly by technical batch, while cell identities and meaningful variation remain distinguishable. scIB provides metrics for these two objectives. Open Problems publishes a concrete batch-integration task; that task and the scIB software are resources, not interchangeable protocol identities.","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"}],"facts":[{"label":"Record type","value":"Generic biological task","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"},{"label":"Inputs","value":"Annotated single-cell data with batch labels and biological information","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"},{"label":"Assessment","value":"Separate biological-conservation and batch-removal assessments","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"},{"label":"Biological conservation examples","value":"Cell-type silhouette, ARI/NMI, isolated-label and trajectory conservation metrics","source_ids":["src-discovery-theislab-scib"],"source_locator":"scIB README: Metrics / Biological Conservation"},{"label":"Batch-removal examples","value":"Batch silhouette, graph iLISI, kBET and graph connectivity","source_ids":["src-discovery-theislab-scib"],"source_locator":"scIB README: Metrics / Batch Correction"},{"label":"Comparator guidance","value":"scIB documents established integration methods including Harmony, MNN and scVI; the choice must match the task inputs.","source_ids":["src-discovery-theislab-scib"],"source_locator":"scIB README: Integration Tools"}],"strengths":[{"text":"A two-part assessment exposes overcorrection that a batch-mixing score alone would miss.","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"}],"limitations":[{"text":"This task record has no single dataset or executable split. No run is implied by its relationship to scIB or Open Problems.","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"}],"diagram":{"title":"Procedure overview","steps":["Annotated cell data","Apply integration method","Check retained biology","Check reduced batch effects"],"caption":"Conceptual overview, not an executable specification.","source_ids":["src-discovery-theislab-scib","src-discovery-openproblems-bio-openproblems"],"source_locator":"scIB README: Metrics; Resources; Open Problems benchmark directory: Batch Integration (https://openproblems.bio/benchmarks/)"},"coverage":"reviewed","gaps":["Select a concrete dataset, integration output type and metric implementation for any future evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}}} {"id":"catalog-task-cell-perturbation","kind":"benchmark","name":"Perturbation response","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Perturbation response","scope_note":"Predict expression changes after unseen perturbations.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"How does gene expression change after an intervention?","sections":[{"title":"Proposed comparison design","body":"A proposed evaluation must specify perturbation identity, cell context, dose, time and which perturbations or contexts are withheld.","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict expression changes after unseen perturbations.","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"No-change and observed-control responses are useful candidate references, not measured results here.","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"A mean expression prediction can hide differential responses; a dataset and metric protocol are still required.","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Define cell context and intervention","Withhold perturbations or contexts","Predict expression response","Compare measured response"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-geneformer","catalog-source-scfoundation","catalog-source-gears"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-cell-reference-mapping","kind":"benchmark","name":"Donor-held-out reference mapping","description":"","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Donor-held-out reference mapping","scope_note":"Map unseen donors to a labelled cell-type reference.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can cells from an unseen donor be assigned to the correct reference cell types?","sections":[{"title":"Proposed comparison design","body":"Build a labelled reference and keep query donors outside downstream fitting. Record whether novel cell types can be rejected instead of forced into a known label.","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Map unseen donors to a labelled cell-type reference.","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Reference-only nearest-neighbour mapping is a candidate comparison when the input representation is matched.","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Randomly withholding cells from the same donor would answer a different generalization question.","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Label reference cells","Withhold query donors","Map query cells to reference","Check labels and unknown types"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-geneformer","catalog-source-scgpt","catalog-source-scfoundation","catalog-source-scvi"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-community-profiling","kind":"benchmark","name":"Community profiling","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-metaphlan"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Community profiling","scope_note":"Estimate taxon abundances in metagenomic samples.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Which microbial taxa occur in a sample, and at what abundance?","sections":[{"title":"Proposed comparison design","body":"A comparison needs matched sample definitions, reference taxonomy versions and a declared abundance convention.","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Estimate taxon abundances in metagenomic samples.","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Taxon presence and abundance errors should be inspected separately.","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Relative abundances depend on the reference and measurement process; this record does not define one mock community.","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Define community samples","Pin reference taxonomy","Estimate taxon abundances","Compare presence and abundance"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-metaphlan"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-complex-structure","kind":"benchmark","name":"Biomolecular complex structure","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Biomolecular complex structure","scope_note":"Predict joint structure for interacting proteins and other molecules.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"What is the joint three-dimensional arrangement of interacting molecules?","sections":[{"title":"Proposed comparison design","body":"Specify all chains and molecular partners, permitted templates and alignments, sampling budget and the rule for choosing the reported structure.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict joint structure for interacting proteins and other molecules.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Interface and whole-complex assessments describe different aspects of structural quality.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"An accurate monomer fold does not establish that its interaction interface or binding partner is correct.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Specify molecular partners","Fix templates and alignments","Predict and select complex","Assess fold and interface"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-alphafold-3-server"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-enhancer-effects","kind":"benchmark","name":"Enhancer / MPRA effects","description":"","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Enhancer / MPRA effects","scope_note":"Predict measured activity changes from regulatory sequence variants.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Do regulatory variants change measured enhancer activity?","sections":[{"title":"Proposed comparison design","body":"Select a specific reporter or endogenous assay, pair reference and alternate sequences and withhold the relevant loci or experimental groups.","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict measured activity changes from regulatory sequence variants.","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"A measured activity endpoint can be compared with simple sequence-based references.","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Reporter activity and endogenous regulation have different contexts; neither implies a universal clinical interpretation.","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose regulatory assay","Pair reference and variant","Predict activity change","Compare measured effects"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-heldout-clade","kind":"benchmark","name":"Held-out-clade classification","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Held-out-clade classification","scope_note":"Hold clades out of downstream fitting and reference databases; evaluate known ancestor labels or unknown-taxon detection, and audit pretraining overlap separately.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can a sequence method generalize beyond taxa present during fitting?","sections":[{"title":"Proposed comparison design","body":"Withhold the chosen clades from downstream training and reference databases, then separately audit overlap with model pretraining.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Hold clades out of downstream fitting and reference databases; evaluate known ancestor labels or unknown-taxon detection, and audit pretraining overlap separately.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Taxonomic distance makes the extrapolation challenge explicit.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Withholding species is not the same as withholding their genera; pretraining overlap is a separate unresolved question.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose withheld clades","Audit training and references","Classify held-out sequences","Assess ancestral or unknown labels"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-evo-2","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-ligand-affinity","kind":"benchmark","name":"Small-molecule affinity","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-boltz-2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Small-molecule affinity","scope_note":"Predict measured binding affinity; pose confidence is not an affinity value.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"How strongly does a molecule bind its target under the measured conditions?","sections":[{"title":"Proposed comparison design","body":"Select a defined affinity assay and unit, separate compounds and targets according to the intended transfer test, and record any structural information supplied.","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict measured binding affinity; pose confidence is not an affinity value.","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Measured affinities provide a quantitative endpoint when assays are comparable.","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"A pose-confidence score is not an affinity; Kd, Ki and activity measurements must not be silently combined.","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose affinity assay","Define compound and target split","Predict binding strength","Compare matched affinity units"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-boltz-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-ligand-pose","kind":"benchmark","name":"Protein–ligand pose","description":"","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Protein–ligand pose","scope_note":"Predict the bound ligand geometry from prepared molecular inputs.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Where and how does a ligand sit in a protein binding site?","sections":[{"title":"Proposed comparison design","body":"Declare whether the pocket is known, how the receptor and ligand are prepared, and how one pose is selected from sampled candidates.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict the bound ligand geometry from prepared molecular inputs.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Geometry validity and agreement with a reference pose capture complementary failure modes.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"A best-of-many pose chosen using the reference cannot be compared with a prospectively selected top-ranked pose.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Prepare receptor and ligand","Declare pocket information","Generate and select poses","Assess geometry and validity"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-chai-1","catalog-source-boltz-2","catalog-source-diffdock-l","catalog-source-alphafold-3-server","catalog-source-vina"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-long-range-regulation","kind":"benchmark","name":"Long-range regulation","description":"","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Long-range regulation","scope_note":"Predict gene-expression or chromatin effects from long-context DNA sequence.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can extended genomic context help predict regulatory effects?","sections":[{"title":"Proposed comparison design","body":"A protocol must identify the genome assembly, genomic interval, input length and measured expression or chromatin endpoint.","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict gene-expression or chromatin effects from long-context DNA sequence.","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Explicit context length makes information available to each model inspectable.","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Different context windows and experimental targets can turn superficially similar scores into different tasks.","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Pin genome and sequence window","Define held-out loci","Predict regulatory endpoint","Compare experimental signal"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-evo-2","catalog-source-alphagenome"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-mfass-splice","kind":"benchmark","name":"MFASS splice-variant prioritisation","description":"","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":["catalog-source-dnabert-2","catalog-source-nt-v2","catalog-source-spliceai","catalog-source-pangolin"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"MFASS splice-variant prioritisation","scope_note":"Functional exon-recognition assay; mfass-v2 reports a corrected baseline and one complete local DNABERT-2 protocol.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"MFASS prioritisation tests whether model scores enrich for experimentally disrupted exon recognition.","sections":[{"title":"Procedure","body":"The concrete rewire v2 protocol validates assay-oriented sequence pairs and evaluates a predeclared grouped split. This catalogue task describes the capability; it is distinct from the assay dataset and the pinned v2 implementation.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"facts":[{"label":"Record type","value":"Task linked to a concrete protocol","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},{"label":"Inputs","value":"Functional reporter labels and variant scores","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},{"label":"Assessment","value":"Top-ranked assay positives and ranking metrics","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"strengths":[{"text":"Experimental reporter labels offer a functional endpoint independent of clinical assertions.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"limitations":[{"text":"Reporter disruption is not a clinical diagnosis, and models with different sequence context are not identical-input baselines.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose matched biological data","Define held-out evaluation","Apply candidate methods","Assess the specified endpoint"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},"coverage":"reviewed","gaps":["Use the pinned MFASS v2 protocol rather than this generic task identity when recording a new run."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}}} {"id":"catalog-task-microbial-promoters","kind":"benchmark","name":"Bacterial promoter prediction","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Bacterial promoter prediction","scope_note":"Classify promoter activity from microbial DNA sequence.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Does a microbial DNA region act as a promoter in the specified organism and assay?","sections":[{"title":"Proposed comparison design","body":"Define the organism, promoter class, negative sampling strategy and held-out sequence groups before evaluating a classifier.","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Classify promoter activity from microbial DNA sequence.","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"A promoter assay anchors a sequence prediction to a specific regulatory function.","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Negative construction and sequence similarity can dominate classification difficulty; no specific split is established here.","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Specify organism and promoter class","Define matched negatives","Predict promoter labels","Score held-out sequence groups"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-prokbert","catalog-source-evo-2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-phage-pathogen-reads","kind":"benchmark","name":"Phage / pathogen reads","description":"","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Phage / pathogen reads","scope_note":"Classify held-out phage or pathogen sequences and record taxonomic distance.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can a method recognize phage or pathogen sequences beyond its references?","sections":[{"title":"Proposed comparison design","body":"Record sequence length, sequencing error, taxonomic distance and reference-database cutoff, then assess the intended classification level.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Classify held-out phage or pathogen sequences and record taxonomic distance.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Held-out taxa allow the evaluation to distinguish recognition from close-reference matching.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"A positive sequence classification does not establish abundance, infectivity or clinical disease.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Define reads and reference cutoff","Withhold target taxa","Classify sequence fragments","Report rank and coverage"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-prokbert","catalog-source-metagene-1","catalog-source-kraken2"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-protein-design","kind":"benchmark","name":"Protein design / inverse folding","description":"","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":["catalog-source-proteinmpnn"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Protein design / inverse folding","scope_note":"Score or design sequences conditional on a known structure.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can a method propose sequences compatible with a desired protein structure?","sections":[{"title":"Proposed comparison design","body":"Separate sequence scoring from generation, define the target structure and permitted constraints, and record how designs are selected for evaluation.","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Score or design sequences conditional on a known structure.","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Structure-conditioned design makes the intended molecular target explicit.","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Sequence recovery or predicted refolding does not establish experimentally measured function.","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Specify target structure","Generate candidate sequences","Select designs without test leakage","Evaluate intended property"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-proteinmpnn"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-protein-monomer-structure","kind":"benchmark","name":"Monomer structure","description":"","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Monomer structure","scope_note":"Predict single-chain structure from sequence.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can a protein sequence be mapped to an accurate single-chain structure?","sections":[{"title":"Proposed comparison design","body":"Specify permitted templates, sequence alignments and structure-release cutoff, then compare predictions with held-out experimental structures.","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict single-chain structure from sequence.","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"An explicit reference structure permits local and global geometric assessment.","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Single-chain accuracy does not imply correct complexes, dynamics or ligand interactions.","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Specify sequence and allowed inputs","Predict single-chain structure","Match experimental reference","Assess local and global geometry"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-esm-2","catalog-source-esmfold","catalog-source-chai-1"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-proteingym-effects","kind":"benchmark","name":"ProteinGym mutation effects","description":"","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":["catalog-source-mimic","catalog-source-esm-2","catalog-source-proteinmpnn"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"ProteinGym mutation effects","scope_note":"Rank substitution effects within held-out deep-mutational-scanning assays.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Protein mutation-effect evaluation asks whether scores order variants consistently with measured assay effects.","sections":[{"title":"Procedure","body":"ProteinGym provides concrete assay releases and evaluation tracks. Select the assay, permitted model inputs and supervision regime before using its scoring and aggregation procedure.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"}],"facts":[{"label":"Record type","value":"Generic task linked to a suite","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"},{"label":"Inputs","value":"Protein variants and experimental assay measurements","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"},{"label":"Assessment","value":"Within-assay ranking and protocol-specific aggregation","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"}],"strengths":[{"text":"Assay-level analysis distinguishes effects measured under different biological conditions.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"}],"limitations":[{"text":"This generic task does not fix one ProteinGym release or train/test split.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose matched biological data","Define held-out evaluation","Apply candidate methods","Assess the specified endpoint"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"ProteinGym README: Overview; Results; Resources"},"coverage":"reviewed","gaps":["Select a ProteinGym release, track and assay subset."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}}} {"id":"catalog-task-rna-secondary-structure","kind":"benchmark","name":"RNA secondary structure","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"RNA secondary structure","scope_note":"Compare predicted base pairs against held-out RNA structures.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Which nucleotides pair within an RNA molecule?","sections":[{"title":"Proposed comparison design","body":"A protocol needs labelled base pairs, a held-out sequence or family split and explicit treatment of pseudoknots and invalid predicted pairs.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Compare predicted base pairs against held-out RNA structures.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Base-pair assessment can identify errors hidden by sequence-level summaries.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"A secondary-structure score does not measure the full three-dimensional fold; RNA-family overlap must be assessed.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Choose RNA sequences and labels","Withhold sequences or families","Predict base pairs","Assess allowed pairing classes"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-rna-splice-sites","kind":"benchmark","name":"RNA splice-site mapping","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"RNA splice-site mapping","scope_note":"Predict splice-site classes from transcript sequence, using a held-out gene split.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Where are splice donor and acceptor sites in a sequence?","sections":[{"title":"Proposed comparison design","body":"Define the sequence convention, splice-site labels and a held-out gene split before fitting and evaluating the predictor.","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict splice-site classes from transcript sequence, using a held-out gene split.","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Gene-held-out assessment addresses reuse of closely related transcript context.","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Splice-site recognition and predicting a variant-induced change in splicing are different tasks.","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Define sequence and site labels","Withhold genes","Predict donor and acceptor sites","Score site recognition"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-rna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-rna-tertiary-structure","kind":"benchmark","name":"RNA tertiary structure","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"RNA tertiary structure","scope_note":"Compare predicted 3D folds against independently held-out structures.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can a method predict an RNA molecule’s three-dimensional fold?","sections":[{"title":"Proposed comparison design","body":"Record the permitted sequence, secondary-structure and template information; compare predictions with independently held-out structures.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Compare predicted 3D folds against independently held-out structures.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"Structural references make geometric errors inspectable.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Input secondary structure, templates and family similarity change the difficulty; none is fixed by this generic task.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Specify RNA and allowed inputs","Predict three-dimensional fold","Match held-out structure","Assess geometric agreement"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-rna-fm","catalog-source-rhofold"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"catalog-task-utr-translation","kind":"benchmark","name":"Translation / RNA stability","description":"","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"links":[],"attributes":{"entity_level":"task","version":null,"task":"Translation / RNA stability","scope_note":"Predict measured translation or stability effects; choose UTR or coding-sequence assays to match each model’s input modality.","missing_metadata":{"protocol_version":"not_yet_extracted"},"profile":{"summary":"Can sequence predict measured RNA translation or stability?","sections":[{"title":"Proposed comparison design","body":"Choose the relevant UTR or coding-sequence assay and specify its cellular context, sequence length and experimental split.","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"facts":[{"label":"Record type","value":"Generic task; proposed design","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Inputs","value":"Predict measured translation or stability effects; choose UTR or coding-sequence assays to match each model’s input modality.","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},{"label":"Assessment","value":"No single dataset, split or scoring implementation is fixed by this task record","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"strengths":[{"text":"A matched assay separates translation and stability endpoints.","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"limitations":[{"text":"Ribosome loading, translation efficiency and RNA half-life are different measurements and should not be pooled.","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"}],"diagram":{"title":"Proposed evaluation design","steps":["Select matched translation or stability assay","Define held-out sequences","Predict specific assay endpoint","Compare measured response"],"caption":"Conceptual design only. This generic task has no fixed executable protocol and does not imply a completed run.","source_ids":["catalog-source-mrna-fm","catalog-source-mimic"],"source_locator":"Catalogue task scope and proposed comparison design; concrete source protocol has not yet been extracted"},"coverage":"limited","gaps":["A concrete protocol, dataset release, split manifest and metric implementation must be selected and source-checked.","Candidate comparisons in this guide are proposals, not recorded evaluations."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"cathe2-2025","kind":"source","name":"CATHe2: Enhanced CATH superfamily detection using ProstT5 and structural alphabets","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12631783/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/biomethods/bpaf080","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"713dbfb6ec1cc1aa85c0543eb93aafa0b45b8873df28053b765dd0a1b6d9b563","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12631783/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:38.366Z","legacy_paper":{"id":"cathe2-2025","title":"CATHe2: Enhanced CATH superfamily detection using ProstT5 and structural alphabets","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12631783/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/biomethods/bpaf080","notes":"Numeric result checked against Table 3. in primary full-text XML; journal/source: Biology Methods & Protocols."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cell-dino-2025","kind":"source","name":"Cell-DINO: Self-supervised image-based embeddings for cell fluorescent microscopy","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12826486/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1371/journal.pcbi.1013828","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"12a53a78c70b3033c3351cf7afd4da42ebc98bb3281308f07e71e5baffc153a0","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12826486/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"cell-dino-2025","title":"Cell-DINO: Self-supervised image-based embeddings for cell fluorescent microscopy","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12826486/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: PLOS Computational Biology; PMC ID: PMC12826486. PL column is F1-score reported on a 0–100 scale; Cell-DINO is a vision encoder plus downstream classifier.","doi":"10.1371/journal.pcbi.1013828"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cell2sentence-2024","kind":"source","name":"Cell2Sentence: Teaching Large Language Models the Language of Biology","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11565894/","version":"preprint archived 2024-10-29","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2023.09.11.557287","publication_status":"preprint","year":2024,"artifact_sha256":"e088727d6e04857fccb7033a9b074e1850f775e86e7d2e99e603dde09558ab02","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11565894/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.533640+00:00","legacy_paper":{"id":"cell2sentence-2024","title":"Cell2Sentence: Teaching Large Language Models the Language of Biology","year":2024,"publication_status":"preprint","version":"preprint archived 2024-10-29","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11565894/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC11565894.","doi":"10.1101/2023.09.11.557287"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"claim-b2-2ome-lm-2025","kind":"claim","name":"Reported AUC for 2OMe-LM","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["human RNA 2-prime-O-methylation site prediction"]},"source_ids":["2ome-lm-2025"],"links":[{"relation":"subject","target_id":"b2-2ome-lm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.919","source_locator":"Table 1, 2OMe-LM row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.332Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-antibody-deamidation-plm-2024","kind":"claim","name":"Reported accuracy for ESM-2 650M embeddings + classifier","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["antibody deamidation-site prediction"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[{"relation":"subject","target_id":"b2-antibody-deamidation-plm-2024"}],"attributes":{"field":"attributes.printed_value","value":"0.944","source_locator":"Table 1, Global embeddings only row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.478Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-barcodebert-2026","kind":"claim","name":"Reported accuracy for BarcodeBERT (4–4-4)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["unseen-species genus classification"]},"source_ids":["barcodebert-2026"],"links":[{"relation":"subject","target_id":"b2-barcodebert-2026"}],"attributes":{"field":"attributes.printed_value","value":"78.5","source_locator":"Table 1, BarcodeBERT (4–4-4) row, unseen-species genus-level 1-NN Acc (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558051+00:00","notes":"Resolved the two-level column header: Acc (%) falls under genus-level 1-NN probe of unseen species, not seen-species classification or BIN reconstruction. BarcodeBERT (4–4-4) has 78.5 in this cell."}}} {"id":"claim-b2-birna-bert-2025","kind":"claim","name":"Reported F1 for BiRNA-BERT","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["extremely long RNA species classification"]},"source_ids":["birna-bert-2025"],"links":[{"relation":"subject","target_id":"b2-birna-bert-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.804","source_locator":"Table 2, BiRNA-BERT row, F1 Score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.292Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-cathe2-2025","kind":"claim","name":"Reported F1 for CATHe2 + ProstT5","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["CATH superfamily annotation"]},"source_ids":["cathe2-2025"],"links":[{"relation":"subject","target_id":"b2-cathe2-2025"}],"attributes":{"field":"attributes.printed_value","value":"82.3","source_locator":"Table 3, ProstT5 full row, F1 score column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:38.366Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-clathrin-plm-2025","kind":"claim","name":"Reported accuracy for ESM-2 embedding + paper classifier","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["clathrin protein classification"]},"source_ids":["clathrin-plm-2025"],"links":[{"relation":"subject","target_id":"b2-clathrin-plm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.916","source_locator":"Table 2, Independent test / ESM-2 row, ACC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558194+00:00","notes":"Resolved the blank evaluation-strategy cells by their independent-test row group. ESM-2 ACC is 0.916 there; the cross-validation ESM-2 ACC is instead 0.873. This is the paper classifier using embeddings, not a standalone checkpoint."}}} {"id":"claim-b2-cobra-rna-binding-2026","kind":"claim","name":"Reported MCC for ERNIE-RNA + CoBRA","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA compound-binding site prediction"]},"source_ids":["cobra-rna-binding-2026"],"links":[{"relation":"subject","target_id":"b2-cobra-rna-binding-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.657","source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558197+00:00","notes":"Matched ERNIE-RNA jointly with TCL focal loss, then the MCC column. Table 2 explicitly reports test-set models. The cell is 0.657, distinct from AUROC 0.868."}}} {"id":"claim-b2-codonbert-vaccines-2024","kind":"claim","name":"Reported Spearman rho for CodonBERT","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["flu-vaccine mRNA property prediction"]},"source_ids":["codonbert-vaccines-2024"],"links":[{"relation":"subject","target_id":"b2-codonbert-vaccines-2024"}],"attributes":{"field":"attributes.printed_value","value":"0.81","source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558201+00:00","notes":"Matched the CodonBERT row and Flu vaccines column (0.81). The table footnote identifies regression columns as Spearman rank correlation and singles out E. coli as classification; this is not a flu-vaccine accuracy score."}}} {"id":"claim-b2-dart-eval-regulatory-2024","kind":"claim","name":"Reported accuracy for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory element identification"]},"source_ids":["dart-eval-regulatory-2024"],"links":[{"relation":"subject","target_id":"b2-dart-eval-regulatory-2024"}],"attributes":{"field":"attributes.printed_value","value":"0.876","source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558203+00:00","notes":"Inspected the pinned NeurIPS primary PDF table and explanatory text. The DNABERT-2 row reports 0.876 under Zero-Shot Accuracy. The caption defines this as pairwise prioritization of positives over matched controls, distinct from supervised absolute accuracy."}}} {"id":"claim-b2-dnabert2-enhancer-2025","kind":"claim","name":"Reported AUC for DNABERT2-Enhancer","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer recognition"]},"source_ids":["dnabert2-enhancer-2025"],"links":[{"relation":"subject","target_id":"b2-dnabert2-enhancer-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.965","source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558204+00:00","notes":"Resolved the first-layer row group. DNABERT2-Enhancer AUC is 0.965, whereas second-layer AUC is 0.933. The caption explicitly describes 5-fold cross-validation on Liu training data, not an independent held-out test."}}} {"id":"claim-b2-eden-genomic-classification-2026","kind":"claim","name":"Reported MCC for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["human core-promoter classification"]},"source_ids":["eden-genomic-classification-2026"],"links":[{"relation":"subject","target_id":"b2-eden-genomic-classification-2026"}],"attributes":{"field":"attributes.printed_value","value":"70.52","source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:37.531Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-b2-ernie-rna-2025","kind":"claim","name":"Reported binary F1 for ERNIE-RNA","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary-structure prediction"]},"source_ids":["ernie-rna-2025"],"links":[{"relation":"subject","target_id":"b2-ernie-rna-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.575","source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558206+00:00","notes":"Resolved bpRNA-new as the first three-column dataset group and F1-Score (binary) as its third metric. ERNIE-RNA zero shot is 86M and reports 0.575; RNA3DB-2D F1 is instead 0.542."}}} {"id":"claim-b2-esm2-ofs-fitness-2025","kind":"claim","name":"Reported Spearman rho for ESM2 OFS pseudo-perplexity","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant fitness prediction"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[{"relation":"subject","target_id":"b2-esm2-ofs-fitness-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.403","source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Publisher PDF retrieved through official APS harvest endpoint after direct download returned403. Table I is ProteinGym substitutions, not indels TableII. Last column aggregate mean0.403; separate function categories precede it. This verifies reported score, not experimental reproduction. Comparator rows in this table are sourced from ProteinGym; OFS PP is authors own method."}}} {"id":"claim-b2-fusion-breakpoint-foundation-models-2026","kind":"claim","name":"Reported ROC AUC for Nucleotide Transformer + NN (middle)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["gene fusion breakpoint classification"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[{"relation":"subject","target_id":"b2-fusion-breakpoint-foundation-models-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.994","source_locator":"Table 2, NT / NN (middle) row, ROC AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558209+00:00","notes":"Matched NT jointly with NN (middle) and ROC AUC 0.994 in the full-test-set table. NT with SVM reports 0.995 and is a separate pipeline."}}} {"id":"claim-b2-genomic-tokenizer-selection-2025","kind":"claim","name":"Reported MCC for Caduceus (character tokens)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory sequence classification"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[{"relation":"subject","target_id":"b2-genomic-tokenizer-selection-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.778","source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558210+00:00","notes":"Matched Regulatory row with Caduceus (char) column, 0.778. Caption establishes these as MCC summaries by category; model-size row identifies 3.9M parameters. This is an aggregated category result, not a single unspecified split."}}} {"id":"claim-b2-gsmformer-ppi-2026","kind":"claim","name":"Reported AUROC for GSMFormer-PPI + ProstT5","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein interaction prediction"]},"source_ids":["gsmformer-ppi-2026"],"links":[{"relation":"subject","target_id":"b2-gsmformer-ppi-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.988","source_locator":"Table 6, ProstT5 embedding row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558212+00:00","notes":"Matched ProstT5 embedding row and AUROC column, 0.988. Caption explicitly describes GSMFormer-PPI using embeddings as node features, not standalone ProstT5 prediction."}}} {"id":"claim-b2-megsite-2025","kind":"claim","name":"Reported AUC for MegSite + ESM3","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["DNA-binding residue prediction"]},"source_ids":["megsite-2025"],"links":[{"relation":"subject","target_id":"b2-megsite-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.948","source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558213+00:00","notes":"Resolved DNA-129_Test row group and ESM3 row. AUC is 0.948; the next numeric cell 0.582 is AP. Caption states an embedding comparison within MegSite."}}} {"id":"claim-b2-mrna-lm-2025","kind":"claim","name":"Reported Spearman rho for mRNA-LM","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA half-life prediction"]},"source_ids":["mrna-lm-2025"],"links":[{"relation":"subject","target_id":"b2-mrna-lm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.696","source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558214+00:00","notes":"Resolved mRNA half-life column under the Spearman header spanning three tasks. mRNA-LM gives 0.696. Caption identifies average test performance across cross-validation splits; protein-expression AUROC is a different column."}}} {"id":"claim-b2-mrnabert-2025","kind":"claim","name":"Reported R-squared for mRNABERT","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["translation-efficiency prediction"]},"source_ids":["mrnabert-2025"],"links":[{"relation":"subject","target_id":"b2-mrnabert-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.669","source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558216+00:00","notes":"Resolved Human group and its R-squared subcolumn. mRNABERT (3066) reports 0.669; Human Spearman is 0.814 and Mouse R-squared is 0.649. Caption specifies ultra-long mRNA translation-efficiency prediction."}}} {"id":"claim-b2-mulan-2025","kind":"claim","name":"Reported AUC for MULAN-ESM2 S","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["human protein-protein interaction prediction"]},"source_ids":["mulan-2025"],"links":[{"relation":"subject","target_id":"b2-mulan-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.717","source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558217+00:00","notes":"Resolved the multirow header: HumanPPI uses AUC. MULAN-ESM2 S has 0.717; this is the small-model group, distinct from M and L variants."}}} {"id":"claim-b2-phylogpn-2025","kind":"claim","name":"Reported AUROC for PhyloGPN","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["ClinVar 3-prime UTR variant classification"]},"source_ids":["phylogpn-2025"],"links":[{"relation":"subject","target_id":"b2-phylogpn-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.94","source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558218+00:00","notes":"Matched 3-prime UTR row and PhyloGPN column (0.94). Caption specifies log-likelihood-ratio predictions of ClinVar classes and explicitly defines each cell as AUROC."}}} {"id":"claim-b2-polya-glm-2025","kind":"claim","name":"Reported AUC for HyenaDNA","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["polyadenylation site detection"]},"source_ids":["polya-glm-2025"],"links":[{"relation":"subject","target_id":"b2-polya-glm-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.7510","source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558220+00:00","notes":"Resolved Few-shot group, HyenaDNA row, and G-G subcolumn under AUC (0.7510). IG-G AUC is 0.7541. Caption states averages over five-fold cross-validation and distinguishes negative sampling regions."}}} {"id":"claim-b2-rlsite-rna-binding-2025","kind":"claim","name":"Reported AUC for RLsite","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA-small-molecule binding-site prediction"]},"source_ids":["rlsite-rna-binding-2025"],"links":[{"relation":"subject","target_id":"b2-rlsite-rna-binding-2025"}],"attributes":{"field":"attributes.printed_value","value":"0.828","source_locator":"Table 1, RLsite row, T18 AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558222+00:00","notes":"Matched RLsite and AUC (0.828). Caption explicitly identifies dataset T18; MCC 0.474 is a different metric."}}} {"id":"claim-b2-rnaret-2026","kind":"claim","name":"Reported F1 for RNAret","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["miRNA-mRNA interaction prediction"]},"source_ids":["rnaret-2026"],"links":[{"relation":"subject","target_id":"b2-rnaret-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.9622","source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558224+00:00","notes":"Resolved the MirTarRAW section, 5-mer RNAret row, and F1 column (0.9622), distinct from DeepMirTarLeft F1 0.9728. Methods confirm 72/8/20 train/validation/test partition for MirTarRAW."}}} {"id":"claim-b2-spin-protein-function-2026","kind":"claim","name":"Reported F1 macro-weighted for SPIN + ESM2-35M","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein function annotation"]},"source_ids":["spin-protein-function-2026"],"links":[{"relation":"subject","target_id":"b2-spin-protein-function-2026"}],"attributes":{"field":"attributes.printed_value","value":"0.796","source_locator":"Table 1, ESM2-35M Test row, F1_m-w column","review":{"method":"independent_ai_table_review","reviewer":"Codex secondary table review","reviewed_at":"2026-09-16T10:38:57.558225+00:00","notes":"Resolved Test group and macro-weighted F1 subcolumn (0.796) for frozen ESM2-35M in SPIN. Test weighted accuracy is 0.798. Methods define inverse-frequency class weighting for macro-weighted F1."}}} {"id":"claim-b2-structure-informed-plm-2025","kind":"claim","name":"Reported AUROC for structure-informed pLM","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein variant-effect classification"]},"source_ids":["structure-informed-plm-2025"],"links":[{"relation":"subject","target_id":"b2-structure-informed-plm-2025"}],"attributes":{"field":"attributes.printed_value","value":".803","source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source review, not human review","reviewed_at":"2026-09-16T10:45:41.099916+00:00","notes":"Full-text HTML succeeds although EuropePMC XMLreturned404. Row is mutation-site variables AA+SS+RSA+CM, not neighbouring environment variant. AUROC .803 is numerically equivalent to preserved legacy0.803. Source check, not experimental reproduction; do not claim original source printed leading zero."}}} {"id":"claim-lit-001","kind":"claim","name":"Reported AUC for Caduceus-Ph","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"subject","target_id":"lit-001"}],"attributes":{"field":"attributes.printed_value","value":"0.783","source_locator":"Table 3, Human 5mC row, Caduceus-Ph column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-002","kind":"claim","name":"Reported AUC for NT-v2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"subject","target_id":"lit-002"}],"attributes":{"field":"attributes.printed_value","value":"0.7377","source_locator":"Table 3, Human 5mC row, NT-v2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-003","kind":"claim","name":"Reported Accuracy for ENBED","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"subject","target_id":"lit-003"}],"attributes":{"field":"attributes.printed_value","value":"90.3","source_locator":"Table 2, Mouse Enhancers row, ENBED column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-004","kind":"claim","name":"Reported Accuracy for ENBED (GRCh38)","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"subject","target_id":"lit-004"}],"attributes":{"field":"attributes.printed_value","value":"81.1","source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-005","kind":"claim","name":"Reported Accuracy for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"subject","target_id":"lit-005"}],"attributes":{"field":"attributes.printed_value","value":"97.0","source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-006","kind":"claim","name":"Reported Accuracy for Caduceus","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"subject","target_id":"lit-006"}],"attributes":{"field":"attributes.printed_value","value":"95.0","source_locator":"Table 5, Caduceus (8 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-007","kind":"claim","name":"Reported AUROC for HyenaDNA","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"subject","target_id":"lit-007"}],"attributes":{"field":"attributes.printed_value","value":"0.828","source_locator":"Table 3, HyenaDNA row, ETGP column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.492545+00:00","notes":"ETGP is the first numeric column, separate from the six CMP cell-type columns and average; caption defines ETGP AUROC. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-008","kind":"claim","name":"Reported AUROC for Caduceus-Ph","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"subject","target_id":"lit-008"}],"attributes":{"field":"attributes.printed_value","value":"0.826","source_locator":"Table 3, Caduceus-Ph row, ETGP column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.493765+00:00","notes":"ETGP is the first numeric column, separate from the six CMP cell-type columns and average; caption defines ETGP AUROC. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-009","kind":"claim","name":"Reported Pearson R for RiNALMo","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"subject","target_id":"lit-009"}],"attributes":{"field":"attributes.printed_value","value":"0.74","source_locator":"Table 2, RiNALMo row, MRL MPRA column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.497221+00:00","notes":"MRL MPRA is the second task under Local and uses R. Caption specifies mean over ten random seeds and selected best model per family, not a fully identified checkpoint. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-010","kind":"claim","name":"Reported Pearson R for RNA-FM","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"subject","target_id":"lit-010"}],"attributes":{"field":"attributes.printed_value","value":"0.49","source_locator":"Table 2, RNA-FM row, MRL MPRA column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.500211+00:00","notes":"MRL MPRA is the second task under Local and uses R. Caption specifies mean over ten random seeds and selected best model per family, not a fully identified checkpoint. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-011","kind":"claim","name":"Reported F1 for BPfold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"subject","target_id":"lit-011"}],"attributes":{"field":"attributes.printed_value","value":"0.814","source_locator":"Table 2, BPfold row, PDB F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.502000+00:00","notes":"PDB is the second four-metric block; its F1 is numeric column six, not Rfam F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-012","kind":"claim","name":"Reported F1 for RNAfold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"subject","target_id":"lit-012"}],"attributes":{"field":"attributes.printed_value","value":"0.747","source_locator":"Table 2, RNAfold row, PDB F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.504220+00:00","notes":"PDB is the second four-metric block; its F1 is numeric column six, not Rfam F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-013","kind":"claim","name":"Reported F1 for TU-Fold (aug)","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"subject","target_id":"lit-013"}],"attributes":{"field":"attributes.printed_value","value":"0.947","source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.505799+00:00","notes":"Overall is the first two-metric block. F1 is first numeric column; source cell includes uncertainty after the preserved central value. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-014","kind":"claim","name":"Reported F1 for UFold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"subject","target_id":"lit-014"}],"attributes":{"field":"attributes.printed_value","value":"0.938","source_locator":"Table 2, UFold row, Overall F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.507031+00:00","notes":"Overall is the first two-metric block. F1 is first numeric column; source cell includes uncertainty after the preserved central value. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-015","kind":"claim","name":"Reported Median F1 for DEBFold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"subject","target_id":"lit-015"}],"attributes":{"field":"attributes.printed_value","value":"55.7","source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.509290+00:00","notes":"TestSet beta is the second four-column block. F1 (%) is its first column; caption reports test-set median F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-016","kind":"claim","name":"Reported Median F1 for RNAfold","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"subject","target_id":"lit-016"}],"attributes":{"field":"attributes.printed_value","value":"52.3","source_locator":"Table 1, RNAfold row, TestSetβ F1 (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.511166+00:00","notes":"TestSet beta is the second four-column block. F1 (%) is its first column; caption reports test-set median F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-017","kind":"claim","name":"Reported Mean Spearman rho for ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"subject","target_id":"lit-017"}],"attributes":{"field":"attributes.printed_value","value":"0.488","source_locator":"Table A7, ESM-2 (15B) row, Stability column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.517323+00:00","notes":"Table A7 is zero-shot substitution DMS grouped by function. Stability is the fifth numeric column; model-type row spans do not change its placement. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-018","kind":"claim","name":"Reported Mean Spearman rho for ProteinMPNN","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"subject","target_id":"lit-018"}],"attributes":{"field":"attributes.printed_value","value":"0.566","source_locator":"Table A7, ProteinMPNN row, Stability column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.523422+00:00","notes":"Table A7 is zero-shot substitution DMS grouped by function. Stability is the fifth numeric column; model-type row spans do not change its placement. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-019","kind":"claim","name":"Reported AUROC for FUJISAN","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"subject","target_id":"lit-019"}],"attributes":{"field":"attributes.printed_value","value":"0.9427","source_locator":"Table 1, FUJISAN row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.728Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-020","kind":"claim","name":"Reported AUROC for ESM2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"subject","target_id":"lit-020"}],"attributes":{"field":"attributes.printed_value","value":"0.7991","source_locator":"Table 1, ESM2 row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.728Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-021","kind":"claim","name":"Reported R² for ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"subject","target_id":"lit-021"}],"attributes":{"field":"attributes.printed_value","value":"0.0248","source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.525183+00:00","notes":"Resolved model row spans and Mean/CLS subrows in JATS: selected Mean, not fine-tuned (cross), Position-Stratified Split > Binding > R-squared. Central value agrees; uncertainty is retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-022","kind":"claim","name":"Reported R² for ESM-C","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"subject","target_id":"lit-022"}],"attributes":{"field":"attributes.printed_value","value":"-0.0162","source_locator":"Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.526387+00:00","notes":"Resolved model row spans and Mean/CLS subrows in JATS: selected Mean, not fine-tuned (cross), Position-Stratified Split > Binding > R-squared. Central value agrees; uncertainty is retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-023","kind":"claim","name":"Reported Mean |Spearman rho| for PST","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"subject","target_id":"lit-023"}],"attributes":{"field":"attributes.printed_value","value":"0.501","source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.527633+00:00","notes":"Zero-shot VEP is the last metric group; selected Mean absolute rho, not GO/EC/binding-site metrics. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-024","kind":"claim","name":"Reported Mean |Spearman rho| for ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"subject","target_id":"lit-024"}],"attributes":{"field":"attributes.printed_value","value":"0.489","source_locator":"Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.528656+00:00","notes":"Zero-shot VEP is the last metric group; selected Mean absolute rho, not GO/EC/binding-site metrics. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-025","kind":"claim","name":"Reported F1-Score for scGPT","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"subject","target_id":"lit-025"}],"attributes":{"field":"attributes.printed_value","value":"0.734","source_locator":"Table 2, M.S. / scGPT row, F1-Score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.530269+00:00","notes":"Selected M.S. dataset block, first scGPT/Geneformer occurrences. F1-Score is last column; later dataset blocks deliberately excluded. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-026","kind":"claim","name":"Reported F1-Score for Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"subject","target_id":"lit-026"}],"attributes":{"field":"attributes.printed_value","value":"0.388","source_locator":"Table 2, M.S. / Geneformer row, F1-Score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.531756+00:00","notes":"Selected M.S. dataset block, first scGPT/Geneformer occurrences. F1-Score is last column; later dataset blocks deliberately excluded. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-027","kind":"claim","name":"Reported Partial-label accuracy for C2S (GPT-2 Large)","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"subject","target_id":"lit-027"}],"attributes":{"field":"attributes.printed_value","value":"0.631","source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.533640+00:00","notes":"Read inline small-caps/bold XML in document order, restoring Geneformer and GPT-2 Large labels. Selected Partial label (first block), L1000 > Acc, not AUROC or Full label. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-028","kind":"claim","name":"Reported Partial-label accuracy for Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"subject","target_id":"lit-028"}],"attributes":{"field":"attributes.printed_value","value":"0.419","source_locator":"Table 3, Partial label / Geneformer row, L1000 Acc column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.535220+00:00","notes":"Read inline small-caps/bold XML in document order, restoring Geneformer and GPT-2 Large labels. Selected Partial label (first block), L1000 > Acc, not AUROC or Full label. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-029","kind":"claim","name":"Reported F1 for scGPT","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"subject","target_id":"lit-029"}],"attributes":{"field":"attributes.printed_value","value":"0.550","source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.537541+00:00","notes":"Selected first hPancreas zero-shot block and F1 last column. Caption says some scores are copied from GenePT; this is source checking of the reported table, not independent experimental evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-030","kind":"claim","name":"Reported F1 for Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"subject","target_id":"lit-030"}],"attributes":{"field":"attributes.printed_value","value":"0.270","source_locator":"Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.539630+00:00","notes":"Selected first hPancreas zero-shot block and F1 last column. Caption says some scores are copied from GenePT; this is source checking of the reported table, not independent experimental evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-031","kind":"claim","name":"Reported AUROC for scRegNet (Geneformer backbone)","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"subject","target_id":"lit-031"}],"attributes":{"field":"attributes.printed_value","value":"0.89","source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.541287+00:00","notes":"Read break elements: cells contain AUROC on first line then AUPRC. Selected hESC (first cell type), first line. Caption specifies 500 most-variable genes and 50 independent evaluations. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-032","kind":"claim","name":"Reported AUROC for scRegNet (scBERT backbone)","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"subject","target_id":"lit-032"}],"attributes":{"field":"attributes.printed_value","value":"0.88","source_locator":"Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.542807+00:00","notes":"Read break elements: cells contain AUROC on first line then AUPRC. Selected hESC (first cell type), first line. Caption specifies 500 most-variable genes and 50 independent evaluations. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-033","kind":"claim","name":"Reported Accuracy for ProkBERT-mini","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"subject","target_id":"lit-033"}],"attributes":{"field":"attributes.printed_value","value":"0.87","source_locator":"Table 3, ProkBERT-mini row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.197Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-034","kind":"claim","name":"Reported Accuracy for Promotech","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"subject","target_id":"lit-034"}],"attributes":{"field":"attributes.printed_value","value":"0.71","source_locator":"Table 3, Promotech row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.197Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-035","kind":"claim","name":"Reported Promoter-class F1 for Eco70PromBERT","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"subject","target_id":"lit-035"}],"attributes":{"field":"attributes.printed_value","value":"0.91","source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.544033+00:00","notes":"F1 score is the third two-column group; selected Promoter subcolumn, not AUROC or precision. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-036","kind":"claim","name":"Reported Promoter-class F1 for iPro70-FMWin","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"subject","target_id":"lit-036"}],"attributes":{"field":"attributes.printed_value","value":"0.90","source_locator":"TABLE 3, iPro70-FMWin row, F1 score Promoter column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.544926+00:00","notes":"F1 score is the third two-column group; selected Promoter subcolumn, not AUROC or precision. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-037","kind":"claim","name":"Reported MCC for EVO2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"subject","target_id":"lit-037"}],"attributes":{"field":"attributes.printed_value","value":"0.680","source_locator":"Table 5, EVO2 row, MCC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.240Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-038","kind":"claim","name":"Reported MCC for geNomad","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"subject","target_id":"lit-038"}],"attributes":{"field":"attributes.printed_value","value":"0.794","source_locator":"Table 5, geNomad row, MCC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.240Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-039","kind":"claim","name":"Reported F1 score for NABAS+","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"subject","target_id":"lit-039"}],"attributes":{"field":"attributes.printed_value","value":"0.719","source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.546107+00:00","notes":"Selected Sample19-new explicitly, not Sample19-old, and F1 rather than precision/recall. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-040","kind":"claim","name":"Reported F1 score for MetaPhlAn3","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"subject","target_id":"lit-040"}],"attributes":{"field":"attributes.printed_value","value":"0.753","source_locator":"Table 3, Sample19-new / MetaPhlAn3 row, F1 score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.547023+00:00","notes":"Selected Sample19-new explicitly, not Sample19-old, and F1 rather than precision/recall. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-041","kind":"claim","name":"Reported Success rate, ligand all-atom RMSD <2 Å for Chai-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"subject","target_id":"lit-041"}],"attributes":{"field":"attributes.printed_value","value":"60.7","source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.548973+00:00","notes":"JATS label is bare 2, which caused original parser miss. LiPP N=331 full-set column selected, not N=36 test subset. Caption success is lipid all-atom RMSD <2 Angstrom; PB-valid is a separate table. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-042","kind":"claim","name":"Reported Success rate, ligand all-atom RMSD <2 Å for DiffDock-L","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"subject","target_id":"lit-042"}],"attributes":{"field":"attributes.printed_value","value":"46.8","source_locator":"Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.550691+00:00","notes":"JATS label is bare 2, which caused original parser miss. LiPP N=331 full-set column selected, not N=36 test subset. Caption success is lipid all-atom RMSD <2 Angstrom; PB-valid is a separate table. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-043","kind":"claim","name":"Reported Forward-screening success rate for DiffDock-NMDN","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"subject","target_id":"lit-043"}],"attributes":{"field":"attributes.printed_value","value":"66.7","source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.552697+00:00","notes":"All scoring functions share DiffDock-NMDN poses via rowspan. Selected forward-screening success percentage, not docking pose success or scoring-power correlation. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-044","kind":"claim","name":"Reported Forward-screening success rate for Vina","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"subject","target_id":"lit-044"}],"attributes":{"field":"attributes.printed_value","value":"42.1","source_locator":"Table 2, Vina scoring row, success rate (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.554518+00:00","notes":"All scoring functions share DiffDock-NMDN poses via rowspan. Selected forward-screening success percentage, not docking pose success or scoring-power correlation. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-045","kind":"claim","name":"Reported Median ligand RMSD for Boltz-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"subject","target_id":"lit-045"}],"attributes":{"field":"attributes.printed_value","value":"1.393","source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.555674+00:00","notes":"JATS label is bare 1. Selected Ligand RMSD column in Plinder-L95, not Protein RMSD; retained method row without refinement settings. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-046","kind":"claim","name":"Reported Median ligand RMSD for DiffDock","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"subject","target_id":"lit-046"}],"attributes":{"field":"attributes.printed_value","value":"1.342","source_locator":"Table 1, DiffDock row, Ligand RMSD (Å) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.556572+00:00","notes":"JATS label is bare 1. Selected Ligand RMSD column in Plinder-L95, not Protein RMSD; retained method row without refinement settings. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-047","kind":"claim","name":"Reported Pearson R for Boltz-2","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"subject","target_id":"lit-047"}],"attributes":{"field":"attributes.printed_value","value":"0.800","source_locator":"Table 3, Boltz-2 row, Pearson’s R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.557756+00:00","notes":"JATS label is bare 3. Selected SARS-CoV-2 Mpro potency Pearson R, not MERS-CoV table 2 or Boltz-2-Internal row; uncertainty retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-048","kind":"claim","name":"Reported Pearson R for DiffDock","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"subject","target_id":"lit-048"}],"attributes":{"field":"attributes.printed_value","value":"0.695","source_locator":"Table 3, DiffDock row, Pearson’s R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.558815+00:00","notes":"JATS label is bare 3. Selected SARS-CoV-2 Mpro potency Pearson R, not MERS-CoV table 2 or Boltz-2-Internal row; uncertainty retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction."}}} {"id":"claim-lit-b3-003","kind":"claim","name":"Reported F1 for Mouse-Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"subject","target_id":"lit-b3-003"}],"attributes":{"field":"attributes.printed_value","value":"48.57","source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.392488+00:00","notes":"h/ Thymus row, four human cell types; zero-shot model blocks use Acc then F1, not fine-tuned scores. Ortholog-based conversion evaluated, not mouse cell annotation. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-004","kind":"claim","name":"Reported F1 for Human-Geneformer","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"subject","target_id":"lit-b3-004"}],"attributes":{"field":"attributes.printed_value","value":"74.48","source_locator":"Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.394073+00:00","notes":"h/ Thymus row, four human cell types; zero-shot model blocks use Acc then F1, not fine-tuned scores. Ortholog-based conversion evaluated, not mouse cell annotation. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-005","kind":"claim","name":"Reported F1 for scLLMDA","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"subject","target_id":"lit-b3-005"}],"attributes":{"field":"attributes.printed_value","value":"0.6525","source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.395850+00:00","notes":"First reference/query block MosA1 to WholeBrainA, F1 second column in block; direction of transfer is part of protocol. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-006","kind":"claim","name":"Reported F1 for MINGLE","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"subject","target_id":"lit-b3-006"}],"attributes":{"field":"attributes.printed_value","value":"0.6256","source_locator":"Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.397272+00:00","notes":"First reference/query block MosA1 to WholeBrainA, F1 second column in block; direction of transfer is part of protocol. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-011","kind":"claim","name":"Reported Adjusted Rand Index for GenePT-w","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"subject","target_id":"lit-b3-011"}],"attributes":{"field":"attributes.printed_value","value":"0.54","source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.399274+00:00","notes":"First Cell type row belongs to Aorta, not preceding Phenotype row or later organs. ARI is first in each three-metric method block, not AMI/ASW. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-012","kind":"claim","name":"Reported Adjusted Rand Index for scGPT","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"subject","target_id":"lit-b3-012"}],"attributes":{"field":"attributes.printed_value","value":"0.47","source_locator":"Table 2, Aorta / Cell type row, scGPT ARI column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.402215+00:00","notes":"First Cell type row belongs to Aorta, not preceding Phenotype row or later organs. ARI is first in each three-metric method block, not AMI/ASW. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-013","kind":"claim","name":"Reported Balanced accuracy for Best frozen single-cell foundation model","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"subject","target_id":"lit-b3-013"}],"attributes":{"field":"attributes.printed_value","value":"0.322","source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:44.421Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-014","kind":"claim","name":"Reported Balanced accuracy for Gene-expression PCA","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"subject","target_id":"lit-b3-014"}],"attributes":{"field":"attributes.printed_value","value":"0.384","source_locator":"Table 2, AIDA v2 row, Gene-expr BA column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:44.421Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-015","kind":"claim","name":"Reported Cell-type accuracy for scaLR","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"subject","target_id":"lit-b3-015"}],"attributes":{"field":"attributes.printed_value","value":"0.942","source_locator":"Table 2, scaLR row, Cell type Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.403844+00:00","notes":"PBMCs-BS all-feature/all-sample cell-type accuracy block, not cell-state accuracy or time. Footnote letters on model labels excluded from identity. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-016","kind":"claim","name":"Reported Cell-type accuracy for scVI + scANVI","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"subject","target_id":"lit-b3-016"}],"attributes":{"field":"attributes.printed_value","value":"0.939","source_locator":"Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.405089+00:00","notes":"PBMCs-BS all-feature/all-sample cell-type accuracy block, not cell-state accuracy or time. Footnote letters on model labels excluded from identity. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-017","kind":"claim","name":"Reported AUC for scXDR","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"subject","target_id":"lit-b3-017"}],"attributes":{"field":"attributes.printed_value","value":"0.8248","source_locator":"Table 2, scXDR row, Scenario 2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.056Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-018","kind":"claim","name":"Reported AUC for scVI","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"subject","target_id":"lit-b3-018"}],"attributes":{"field":"attributes.printed_value","value":"0.6970","source_locator":"Table 2, scVI row, Scenario 2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.056Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-019","kind":"claim","name":"Reported L1 abundance error for CAMMiQ","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"subject","target_id":"lit-b3-019"}],"attributes":{"field":"attributes.printed_value","value":"0.0517","source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.408237+00:00","notes":"HumanGut-all in B. L1 Err. block (second occurrence), not strain count or C. L2 Err. Numbers are errors; lower is better. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-020","kind":"claim","name":"Reported L1 abundance error for Kraken2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"subject","target_id":"lit-b3-020"}],"attributes":{"field":"attributes.printed_value","value":"0.2841","source_locator":"Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.411180+00:00","notes":"HumanGut-all in B. L1 Err. block (second occurrence), not strain count or C. L2 Err. Numbers are errors; lower is better. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-021","kind":"claim","name":"Reported Genus-level F1 for Lazypipe-nt","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"subject","target_id":"lit-b3-021"}],"attributes":{"field":"attributes.printed_value","value":"0.932","source_locator":"Table 1, Lazypipe-nt / Genus row, F column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.412605+00:00","notes":"First Genus block selected using rank row span, not Species. Final F column is F score, not precision or recall. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-022","kind":"claim","name":"Reported Genus-level F1 for Kraken2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"subject","target_id":"lit-b3-022"}],"attributes":{"field":"attributes.printed_value","value":"0.627","source_locator":"Table 1, Kraken2 / Genus row, F column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.413635+00:00","notes":"First Genus block selected using rank row span, not Species. Final F column is F score, not precision or recall. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-023","kind":"claim","name":"Reported Macro F1 for NCD-gzip","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["CAMI II superkingdom read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"subject","target_id":"lit-b3-023"}],"attributes":{"field":"attributes.printed_value","value":"0.9804","source_locator":"Table 5, NCD Superkingdom row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.414806+00:00","notes":"NCD rank-specific table 5, F1 column; Superkingdom and Phylum are different classification granularities. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-024","kind":"claim","name":"Reported Macro F1 for NCD-gzip","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["CAMI II phylum read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"subject","target_id":"lit-b3-024"}],"attributes":{"field":"attributes.printed_value","value":"0.1263","source_locator":"Table 5, NCD Phylum row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.415788+00:00","notes":"NCD rank-specific table 5, F1 column; Superkingdom and Phylum are different classification granularities. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-025","kind":"claim","name":"Reported Average prophage F1 for VIBRANT","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"subject","target_id":"lit-b3-025"}],"attributes":{"field":"attributes.printed_value","value":"0.169","source_locator":"Table 3, Vibrant row, Prophage F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.134Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-026","kind":"claim","name":"Reported Average prophage F1 for VirSorter","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"subject","target_id":"lit-b3-026"}],"attributes":{"field":"attributes.printed_value","value":"0.147","source_locator":"Table 3, VirSorter row, Prophage F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.134Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-027","kind":"claim","name":"Reported F1 for GenomeOcean","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"subject","target_id":"lit-b3-027"}],"attributes":{"field":"attributes.printed_value","value":"99.03","source_locator":"Table 2, GenomeOcean row, F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.224Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-028","kind":"claim","name":"Reported F1 for DNABERT-2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"subject","target_id":"lit-b3-028"}],"attributes":{"field":"attributes.printed_value","value":"85.12","source_locator":"Table 2, DNABERT-2 row, F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.224Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-029","kind":"claim","name":"Reported Genus-level F1 for kMetaShot","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"subject","target_id":"lit-b3-029"}],"attributes":{"field":"attributes.printed_value","value":"95.83","source_locator":"Table 2, F1-score % row, Genus kMS column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.417367+00:00","notes":"Real mock sequencing table, F1-score percentage row; Genus is final three-column block, selecting kMS or Gtk rather than Species/Strain. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-030","kind":"claim","name":"Reported Genus-level F1 for GTDB-Tk","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"subject","target_id":"lit-b3-030"}],"attributes":{"field":"attributes.printed_value","value":"89.80","source_locator":"Table 2, F1-score % row, Genus Gtk column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.418914+00:00","notes":"Real mock sequencing table, F1-score percentage row; Genus is final three-column block, selecting kMS or Gtk rather than Species/Strain. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-031","kind":"claim","name":"Reported F1 for Lemur","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"subject","target_id":"lit-b3-031"}],"attributes":{"field":"attributes.printed_value","value":"0.376","source_locator":"Table 3, LOG 10% / Lemur row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.420493+00:00","notes":"LOG 10% first block, F1 column. Kraken 2 row inherits dataset via rowspan; not LOG 75% or abundance Spearman. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-032","kind":"claim","name":"Reported F1 for Kraken 2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"subject","target_id":"lit-b3-032"}],"attributes":{"field":"attributes.printed_value","value":"0.375","source_locator":"Table 3, LOG 10% / Kraken 2 row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.421888+00:00","notes":"LOG 10% first block, F1 column. Kraken 2 row inherits dataset via rowspan; not LOG 75% or abundance Spearman. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-033","kind":"claim","name":"Reported Mean AUC for iPro-MP","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"subject","target_id":"lit-b3-033"}],"attributes":{"field":"attributes.printed_value","value":"0.935","source_locator":"Table 2, iPro-MP row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.361Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-034","kind":"claim","name":"Reported Mean AUC for Prompt","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"subject","target_id":"lit-b3-034"}],"attributes":{"field":"attributes.printed_value","value":"0.835","source_locator":"Table 2, Prompt row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.361Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-035","kind":"claim","name":"Reported Genus macro AveP for ICCTax","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"subject","target_id":"lit-b3-035"}],"attributes":{"field":"attributes.printed_value","value":"67.20","source_locator":"Table 2, ICCTax row, Genus column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.373Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-036","kind":"claim","name":"Reported Genus macro AveP for Kraken2","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"subject","target_id":"lit-b3-036"}],"attributes":{"field":"attributes.printed_value","value":"70.56","source_locator":"Table 2, Kraken2 row, Genus column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.373Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-037","kind":"claim","name":"Reported AUC-ROC for Chai-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"subject","target_id":"lit-b3-037"}],"attributes":{"field":"attributes.printed_value","value":"0.86","source_locator":"Table 5, Folded row, Chai-1 (no MSA) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.400Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-038","kind":"claim","name":"Reported AUC-ROC for Boltz-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"subject","target_id":"lit-b3-038"}],"attributes":{"field":"attributes.printed_value","value":"0.85","source_locator":"Table 5, Folded row, Boltz-1 (no MSA) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.400Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-039","kind":"claim","name":"Reported Top-1 ligand RMSD <2 Å rate for Boltz-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz1-2025"],"links":[{"relation":"subject","target_id":"lit-b3-039"}],"attributes":{"field":"attributes.printed_value","value":"0.545","source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.424864+00:00","notes":"3 recycling rounds and 200 steps; L-RMSD <2 Angstrom top-1 (last column), not oracle. Five samples generated; top-1 means highest-confidence candidate. Repeated reference rows are one evaluation, not independent experiments. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-040","kind":"claim","name":"Reported Mean CDR H3 RMSD for Ibex","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"subject","target_id":"lit-b3-040"}],"attributes":{"field":"attributes.printed_value","value":"2.72","source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.426811+00:00","notes":"First Antibodies block, CDR H3 mean RMSD in Angstrom; excludes later Nanobodies and TCR blocks. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-041","kind":"claim","name":"Reported Mean CDR H3 RMSD for Chai-1","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"subject","target_id":"lit-b3-041"}],"attributes":{"field":"attributes.printed_value","value":"2.65","source_locator":"Table 1, Antibodies / Chai-1 row, CDR H3 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.428536+00:00","notes":"First Antibodies block, CDR H3 mean RMSD in Angstrom; excludes later Nanobodies and TCR blocks. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-042","kind":"claim","name":"Reported Pearson R for DEELIG","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"subject","target_id":"lit-b3-042"}],"attributes":{"field":"attributes.printed_value","value":"0.889","source_locator":"Table 2, DEELIG row, PDBbind v2016 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.586Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-043","kind":"claim","name":"Reported Pearson R for TOPBP (Complex)","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"subject","target_id":"lit-b3-043"}],"attributes":{"field":"attributes.printed_value","value":"0.861","source_locator":"Table 2, TOPBP (Complex) row, PDBbind v2016 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.429669+00:00","notes":"TOPBP Complex reference row; PDBbind v2016 core-set Pearson correlation. Third-party comparator with cited reference; do not infer an independent new run from table inclusion. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-044","kind":"claim","name":"Reported RMSD ≤1 Å and PB-valid success for MolAS","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"subject","target_id":"lit-b3-044"}],"attributes":{"field":"attributes.printed_value","value":"36.69","source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.432565+00:00","notes":"PoseBusters, Mixed, AutoDock row within jointly trained with/without relaxation block. Selected RMSD <=1 Angstrom AND PB-valid group; five-fold average success percentage, not <=2 Angstrom. Inline bold digit nodes joined in original order. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-045","kind":"claim","name":"Reported RMSD ≤1 Å and PB-valid success for Single best solver","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"subject","target_id":"lit-b3-045"}],"attributes":{"field":"attributes.printed_value","value":"34.34","source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, SBS success column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.435454+00:00","notes":"PoseBusters, Mixed, AutoDock row within jointly trained with/without relaxation block. Selected RMSD <=1 Angstrom AND PB-valid group; five-fold average success percentage, not <=2 Angstrom. Inline bold digit nodes joined in original order. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-046","kind":"claim","name":"Reported Docked frames best-matched RMSD <3 Å for AutoDock Vina holo","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"subject","target_id":"lit-b3-046"}],"attributes":{"field":"attributes.printed_value","value":"27.96","source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:56.275Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-047","kind":"claim","name":"Reported Docked frames best-matched RMSD <3 Å for DiffDock holo","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"subject","target_id":"lit-b3-047"}],"attributes":{"field":"attributes.printed_value","value":"21.32","source_locator":"Table 2, Ligand 47 row, DiffDock Holo Docking column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:56.275Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b3-048","kind":"claim","name":"Reported Pearson R for AK-score-ensemble","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"subject","target_id":"lit-b3-048"}],"attributes":{"field":"attributes.printed_value","value":"0.812","source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.436853+00:00","notes":"CASF-2016 scoring Pearson R with learning rate0.0007. Single-model versus ensemble blocks kept distinct; ranking/docking scores not substituted. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-049","kind":"claim","name":"Reported Pearson R for AK-score-single","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"subject","target_id":"lit-b3-049"}],"attributes":{"field":"attributes.printed_value","value":"0.759","source_locator":"Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.437894+00:00","notes":"CASF-2016 scoring Pearson R with learning rate0.0007. Single-model versus ensemble blocks kept distinct; ranking/docking scores not substituted. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-050","kind":"claim","name":"Reported Pearson R for PMF + ECFP + PF (LightGBM)","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"subject","target_id":"lit-b3-050"}],"attributes":{"field":"attributes.printed_value","value":"0.79","source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.439460+00:00","notes":"Resolved two-row model rowspan: last LightGBM row is PMF+ECFP+PF; first LASSO row is PMF. Pearson R, not RMSE. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b3-051","kind":"claim","name":"Reported Pearson R for PMF (LASSO)","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"subject","target_id":"lit-b3-051"}],"attributes":{"field":"attributes.printed_value","value":"0.67","source_locator":"Table 1, PMF / LASSO row, R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.440812+00:00","notes":"Resolved two-row model rowspan: last LightGBM row is PMF+ECFP+PF; first LASSO row is PMF. Pearson R, not RMSE. Source check verifies central value and table context, not experiment reproduction or all metadata."}}} {"id":"claim-lit-b4-001","kind":"claim","name":"Reported AUROC for ARSENAL+ChromBPNet","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory-variant scoring"]},"source_ids":["arsenal-regulatory-dna-2026"],"links":[{"relation":"subject","target_id":"lit-b4-001"}],"attributes":{"field":"attributes.printed_value","value":"0.896","source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Yoruban LCL dsQTLs; ARSENAL+ChromBPNet AUROC Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-002","kind":"claim","name":"Reported AUROC for PlantCAD2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["cross-species conservation prediction"]},"source_ids":["plantcad2-2025"],"links":[{"relation":"subject","target_id":"lit-b4-002"}],"attributes":{"field":"attributes.printed_value","value":"0.725","source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"First comparison entry is PlantCAD2; AUROC is 0.725 versus comparator 0.691. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-003","kind":"claim","name":"Reported accuracy for Stacking-Auto","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["hi-enhancer-2025"],"links":[{"relation":"subject","target_id":"lit-b4-003"}],"attributes":{"field":"attributes.printed_value","value":"80.50","source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Ours row, Accuracy column; original source method is the Stacking-Auto stage. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-004","kind":"claim","name":"Reported AUROC for position-aware CNN","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["enhancer-position-encoding-2024"],"links":[{"relation":"subject","target_id":"lit-b4-004"}],"attributes":{"field":"attributes.printed_value","value":"0.94","source_locator":"Table 2, Human section, CNN row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Human dataset row, CNN, AUC column. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-005","kind":"claim","name":"Reported F1 for ADAR-GPT continual","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["A-to-I RNA editing site prediction"]},"source_ids":["adar-gpt-editing-2026"],"links":[{"relation":"subject","target_id":"lit-b4-005"}],"attributes":{"field":"attributes.printed_value","value":"0.763","source_locator":"Table 2, Adar-GPT (continual) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Adar-GPT continual row; XML inline decimal reordered by parser, original text verified separately. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-006","kind":"claim","name":"Reported sequence recovery for R3Design","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA sequence design"]},"source_ids":["r3design-2025"],"links":[{"relation":"subject","target_id":"lit-b4-006"}],"attributes":{"field":"attributes.printed_value","value":"43.27","source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"R3Design row, first Recovery column Rfam; 43.27 plus/minus0.56. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-007","kind":"claim","name":"Reported AUROC for CUPID Data-aug-Avg","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["non-coding RNA pairwise interaction prediction"]},"source_ids":["cupid-rna-interactions-2026"],"links":[{"relation":"subject","target_id":"lit-b4-007"}],"attributes":{"field":"attributes.printed_value","value":"0.919","source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"CUPID section Data-aug-Avg row; AUROC not AUPRC. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-008","kind":"claim","name":"Reported AUROC for ProteinBERT LLM-encoding model","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA-protein interaction prediction"]},"source_ids":["mrna-protein-diversity-2026"],"links":[{"relation":"subject","target_id":"lit-b4-008"}],"attributes":{"field":"attributes.printed_value","value":"71.5","source_locator":"Table 2, RBP-aware test set row, auROC (%) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.257Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-009","kind":"claim","name":"Reported AUROC for ESM2 650M","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["human-versus-viral protein classification"]},"source_ids":["viral-immune-mimicry-2025"],"links":[{"relation":"subject","target_id":"lit-b4-009"}],"attributes":{"field":"attributes.printed_value","value":"99.67","source_locator":"Table 1, ESM2 650M row, AUC (%) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.274Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-010","kind":"claim","name":"Reported AUROC for ProtT5 embeddings + ensemble classifier","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein binding-site prediction"]},"source_ids":["protein-binding-sites-2023"],"links":[{"relation":"subject","target_id":"lit-b4-010"}],"attributes":{"field":"attributes.printed_value","value":"0.810","source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Dset_448 block; ProtT5 AUROC, downstream ensemble retained in protocol. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-011","kind":"claim","name":"Reported AUROC for CLAPE-SMB with ESM-2","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-small molecule binding-site prediction"]},"source_ids":["clape-smb-2024"],"links":[{"relation":"subject","target_id":"lit-b4-011"}],"attributes":{"field":"attributes.printed_value","value":"0.917","source_locator":"Table 5, ESM-2 / SJC row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"ESM-2 on SJC AUROC. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-012","kind":"claim","name":"Reported AUPRC for Vaxign-DL + ESM","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["vaccine-antigen candidate prediction"]},"source_ids":["vaxign-esm-2024"],"links":[{"relation":"subject","target_id":"lit-b4-012"}],"attributes":{"field":"attributes.printed_value","value":"0.92","source_locator":"Table 2, 4 Layers row, AUPRC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Source spells 4 Layerss; AUPRC0.92±0.013. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-013","kind":"claim","name":"Reported AUROC for scGPT + residual geometry","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["gene-regulatory signal prediction"]},"source_ids":["single-cell-residual-geometry-2026"],"links":[{"relation":"subject","target_id":"lit-b4-013"}],"attributes":{"field":"attributes.printed_value","value":"0.677","source_locator":"Table 4, Immune row, scGPT > +geom AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Immune row, scGPT +geom (second numeric column), not Geneformer or delta. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-014","kind":"claim","name":"Reported F1 for GREmLN","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["cell-type annotation"]},"source_ids":["gremln-2026"],"links":[{"relation":"subject","target_id":"lit-b4-014"}],"attributes":{"field":"attributes.printed_value","value":"0.937","source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.502Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-015","kind":"claim","name":"Reported F1 for Cell-DINO ViT-L","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["protein localization classification"]},"source_ids":["cell-dino-2025"],"links":[{"relation":"subject","target_id":"lit-b4-015"}],"attributes":{"field":"attributes.printed_value","value":"65.5","source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"HPA-FoV Cell-DINO PL column (protein localisation), not CL. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-016","kind":"claim","name":"Reported precision at 50% recall for scGen","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["differentially expressed gene identification"]},"source_ids":["insilico-perturbation-auprc-2025"],"links":[{"relation":"subject","target_id":"lit-b4-016"}],"attributes":{"field":"attributes.printed_value","value":"0.91","source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"CD14+Mono scGen; precision at50%recall. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-017","kind":"claim","name":"Reported F1 for TCINet + HTRS","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["pathogen detection"]},"source_ids":["metagenomic-pathogens-2025"],"links":[{"relation":"subject","target_id":"lit-b4-017"}],"attributes":{"field":"attributes.printed_value","value":"0.84","source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"MetaHIT block TCINet+HTRS F1-score. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-018","kind":"claim","name":"Reported accuracy for DETIRE","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["viral sequence detection"]},"source_ids":["detire-viral-metagenomes-2023"],"links":[{"relation":"subject","target_id":"lit-b4-018"}],"attributes":{"field":"attributes.printed_value","value":"0.8772","source_locator":"Table 1, Accuracy row, DETIRE column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.392Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-019","kind":"claim","name":"Reported accuracy for PC-mer + LR","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["metagenomic genus classification"]},"source_ids":["pc-mer-2024"],"links":[{"relation":"subject","target_id":"lit-b4-019"}],"attributes":{"field":"attributes.printed_value","value":"96.95","source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"AMP block PC-mer+LR section, k=8, first numeric value after k is Accuracy. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-020","kind":"claim","name":"Reported accuracy for MDL4Microbiome","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["microbiome disease-state classification"]},"source_ids":["mdl4microbiome-2022"],"links":[{"relation":"subject","target_id":"lit-b4-020"}],"attributes":{"field":"attributes.printed_value","value":"0.97","source_locator":"Table 3, CRC row, MDL4Microbiome column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.492Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-021","kind":"claim","name":"Reported Pearson correlation for binding-affinity meta-model","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["protein-ligand binding affinity prediction"]},"source_ids":["ligand-affinity-meta-model-2024"],"links":[{"relation":"subject","target_id":"lit-b4-021"}],"attributes":{"field":"attributes.printed_value","value":"0.777","source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Meta-models CASF-2016 PCC. Confirmed XML training-set rowspan inherits preceding row, so0.777 maps to PCC. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-022","kind":"claim","name":"Reported AUROC for DeepInterAware","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["antigen-antibody HIV neutralization prediction"]},"source_ids":["deepinteraware-2025"],"links":[{"relation":"subject","target_id":"lit-b4-022"}],"attributes":{"field":"attributes.printed_value","value":"0.826","source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Ab Unseen block DeepInterAware AUROC0.826±0.017, not Ag Unseen. Transcription verified; experimental claims not independently reproduced."}}} {"id":"claim-lit-b4-023","kind":"claim","name":"Reported AUROC for TransBind","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["transcription-factor DNA binding-site prediction"]},"source_ids":["transbind-2026"],"links":[{"relation":"subject","target_id":"lit-b4-023"}],"attributes":{"field":"attributes.printed_value","value":"0.9508","source_locator":"Table 2, TransBind row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.585Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"claim-lit-b4-024","kind":"claim","name":"Reported AUROC for ESM2_AMPS","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["protein-protein interaction prediction"]},"source_ids":["esm2-amp-2025"],"links":[{"relation":"subject","target_id":"lit-b4-024"}],"attributes":{"field":"attributes.printed_value","value":"0.68","source_locator":"Table 4, ESM2_AMPS row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.625Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness."}}} {"id":"clape-smb-2024","kind":"source","name":"Protein-small molecule binding site prediction based on a pre-trained protein language model with contrastive learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11542454/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1186/s13321-024-00920-2","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"215919244c3dd2dfb0b55fce91c211430fd8d4aee4bb28bd03eab9f4feb73e62","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11542454/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"clape-smb-2024","title":"Protein-small molecule binding site prediction based on a pre-trained protein language model with contrastive learning","year":2024,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11542454/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Journal of Cheminformatics; PMC ID: PMC11542454. ESM-2 feature extractor embedded in CLAPE-SMB; score belongs to combined downstream system.","doi":"10.1186/s13321-024-00920-2"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"clathrin-plm-2025","kind":"source","name":"Advancing the accuracy of clathrin protein prediction through multi-source protein language models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12238356/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-08510-4","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"2edc86b25707c1b737d26117093ce8d856e79cc5d0b335f27c1c341f887f1c7e","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12238356/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558194+00:00","legacy_paper":{"id":"clathrin-plm-2025","title":"Advancing the accuracy of clathrin protein prediction through multi-source protein language models","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12238356/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-08510-4","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Scientific Reports."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cobra-rna-binding-2026","kind":"source","name":"CoBRA: compound binding site prediction using RNA language model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12790621/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bib/bbaf713","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"8c6a6f00f5fa5f62acf301a66e9e6fa9ef11c7a05ad9b7447d2ade2ce8eba793","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12790621/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558197+00:00","legacy_paper":{"id":"cobra-rna-binding-2026","title":"CoBRA: compound binding site prediction using RNA language model","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12790621/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bib/bbaf713","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Briefings in Bioinformatics."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"codonbert-vaccines-2024","kind":"source","name":"CodonBERT large language model for mRNA vaccines","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11368176/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1101/gr.278870.123","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"2968073753e6d44feff9c08b131edf23145e95b171434539dddf77bb92847033","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11368176/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558201+00:00","legacy_paper":{"id":"codonbert-vaccines-2024","title":"CodonBERT large language model for mRNA vaccines","year":2024,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11368176/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1101/gr.278870.123","notes":"Numeric result checked against Table 2. in primary full-text XML; journal/source: Genome Research."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cupid-rna-interactions-2026","kind":"source","name":"Computational understanding of non-coding RNA pairwise interactions","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12957212/","version":"PMC archival version PMC12957212.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.3389/frai.2026.1749205","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"0e6719410b390ee9c4858bb9321042851100fb74df3aa109bf2af2b8aaff7ac1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12957212/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"cupid-rna-interactions-2026","title":"Computational understanding of non-coding RNA pairwise interactions","year":2026,"publication_status":"peer_reviewed","version":"PMC archival version PMC12957212.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12957212/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Frontiers in Artificial Intelligence; PMC ID: PMC12957212. RNA-RNA pairwise interaction predictor; not a foundation model.","doi":"10.3389/frai.2026.1749205"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"cyaprombert-2022","kind":"source","name":"TSSNote-CyaPromBERT: Development of an integrated platform for highly accurate promoter prediction and visualization of Synechococcus sp. and Synechocystis sp. through a state-of-the-art natural language processing model BERT","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9745317/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.3389/fgene.2022.1067562","publication_status":"peer_reviewed","year":2022,"artifact_sha256":"74278ccd77b2bc00a3f4434546545e8bdec8b0652a0e5d1862ec0f91decccd8d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9745317/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.544033+00:00","legacy_paper":{"id":"cyaprombert-2022","title":"TSSNote-CyaPromBERT: Development of an integrated platform for highly accurate promoter prediction and visualization of Synechococcus sp. and Synechocystis sp. through a state-of-the-art natural language processing model BERT","year":2022,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9745317/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Frontiers in Genetics; PMC ID: PMC9745317.","doi":"10.3389/fgene.2022.1067562"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"dart-eval-regulatory-2024","kind":"source","name":"DART-Eval: A Comprehensive DNA Language Model Evaluation Benchmark on Regulatory DNA","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","version":"NeurIPS 2024 Datasets and Benchmarks Track proceedings","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.52202/079017-1981","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"e5aee5b1f7cc6fd961b1d2a131d02cf243b79e091d5e418fbabee7fde9b39b22","artifact_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","artifact_retrieved_at":"2026-09-16T10:38:57.558203+00:00","legacy_paper":{"id":"dart-eval-regulatory-2024","title":"DART-Eval: A Comprehensive DNA Language Model Evaluation Benchmark on Regulatory DNA","year":2024,"publication_status":"peer_reviewed","version":"NeurIPS 2024 Datasets and Benchmarks Track proceedings","source_url":"https://proceedings.neurips.cc/paper_files/paper/2024/file/71998bfc3217ffe1cca1ee084dfadadd-Paper-Datasets_and_Benchmarks_Track.pdf","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","notes":"Proceedings Table 3, DNABERT-2 Zero-Shot Accuracy 0.876 checked directly; the PMC/arXiv manuscript carries the same printed row.","doi":"10.52202/079017-1981"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"debfold-2024","kind":"source","name":"DEBFold: Computational Identification of RNA Secondary Structures for Sequences across Structural Families Using Deep Learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11094721/","version":"PMC11094721.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acs.jcim.4c00458","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"e8f960eafb7f00edfdd81d4fb75c6de838e9b872b7e18875fc7a5bff2a2f72b3","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11094721/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.509290+00:00","legacy_paper":{"id":"debfold-2024","title":"DEBFold: Computational Identification of RNA Secondary Structures for Sequences across Structural Families Using Deep Learning","year":2024,"publication_status":"peer_reviewed","version":"PMC11094721.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11094721/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC11094721.","doi":"10.1021/acs.jcim.4c00458"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"deelig-2021","kind":"source","name":"DEELIG: A Deep Learning Approach to Predict Protein-Ligand Binding Affinity","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8274096/","version":"PMC archival version PMC8274096.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1177/11779322211030364","publication_status":"peer_reviewed","year":2021,"artifact_sha256":"5a7620c18d0622561004e1e25b5cfaf7399e93df3547eeefdd4cf6d300bb8aba","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8274096/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.586Z","legacy_paper":{"id":"deelig-2021","title":"DEELIG: A Deep Learning Approach to Predict Protein-Ligand Binding Affinity","year":2021,"publication_status":"peer_reviewed","version":"PMC archival version PMC8274096.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8274096/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Bioinformatics and Biology Insights; PMC ID: PMC8274096.","doi":"10.1177/11779322211030364"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"deepinteraware-2025","kind":"source","name":"DeepInterAware: Deep Interaction Interface‐Aware Network for Improving Antigen‐Antibody Interaction Prediction from Sequence Data","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11967782/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1002/advs.202412533","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"25d3561934965f754d8712ec02b2052e9a3979e433b88ecebd5db14e930f17a1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11967782/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"deepinteraware-2025","title":"DeepInterAware: Deep Interaction Interface‐Aware Network for Improving Antigen‐Antibody Interaction Prediction from Sequence Data","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11967782/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Advanced Science; PMC ID: PMC11967782. Neutralization prediction, not generic binding affinity; uncertainty printed in source table.","doi":"10.1002/advs.202412533"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"detire-viral-metagenomes-2023","kind":"source","name":"DETIRE: a hybrid deep learning model for identifying viral sequences from metagenomes","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10313334/","version":"PMC archival version PMC10313334.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.3389/fmicb.2023.1169791","publication_status":"peer_reviewed","year":2023,"artifact_sha256":"9ff7d32758620f7b0b0628425f62abff103ca2e33269ce3763383584bcebfc3c","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10313334/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:58.392Z","legacy_paper":{"id":"detire-viral-metagenomes-2023","title":"DETIRE: a hybrid deep learning model for identifying viral sequences from metagenomes","year":2023,"publication_status":"peer_reviewed","version":"PMC archival version PMC10313334.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10313334/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Frontiers in Microbiology; PMC ID: PMC10313334. Task-specific viral classifier, included as a microbial metagenomics benchmark.","doi":"10.3389/fmicb.2023.1169791"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Matched control cells, normalization and evaluation gene set.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-cell-perturbation-no-change","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"Cell perturbation no-change","source_ids":["src-discovery-altoslabs-perturbench"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Receptor preparation, search box, exhaustiveness and conformers.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["molecular-interactions"]},"id":"discovery-baseline-classical-molecular-docking","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-plinder"},{"relation":"model","target_id":"discovery-model-autodock-vina"}],"name":"Classical molecular docking","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Matched sequence lengths and dinucleotide-preserving shuffle; fix seeds.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-dinucleotide-shuffled-sequence-control","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"}],"name":"Dinucleotide-shuffled sequence control","source_ids":["src-discovery-kundajelab-dart-eval"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned reference genomes, taxonomy and confidence setting.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["microbiome"]},"id":"discovery-baseline-exact-sequence-taxonomic-classification","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami-taxonomic-binning"},{"relation":"model","target_id":"discovery-model-kraken-2"}],"name":"Exact-sequence taxonomic classification","source_ids":["src-discovery-derrickwood-kraken2"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"experimental-reference","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Independent biological replicates under matching conditions; not a universal ceiling.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-experimental-replicate-agreement","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-scperteval"}],"name":"Experimental replicate agreement","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"mechanistic","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Stoichiometric reconstruction, growth medium, bounds and objective.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["mechanistic-biology"]},"id":"discovery-baseline-flux-balance-prediction","kind":"baseline","links":[{"relation":"model","target_id":"discovery-model-cobrapy"}],"name":"Flux-balance prediction","source_ids":["src-discovery-opencobra-cobrapy"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only motif features and leakage-aware glycan split.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["glycomics"]},"id":"discovery-baseline-glycan-motif-feature-classifier","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-glycanml"}],"name":"Glycan motif feature classifier","source_ids":["src-discovery-bojarlab-glycowork"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only feature fitting; choose k and penalty within training folds.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-k-mer-ridge-regression","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-genomic-benchmarks"}],"name":"k-mer ridge regression","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Adduct, ion mode, library version, mass tolerance and annotation resolution.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["lipidomics"]},"id":"discovery-baseline-lipid-fragmentation-library-match","kind":"baseline","links":[{"relation":"model","target_id":"discovery-model-lipidblast"}],"name":"Lipid fragmentation library match","source_ids":["src-discovery-lipidblast"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned marker database and taxonomic rank.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["microbiome"]},"id":"discovery-baseline-marker-based-microbial-profiling","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami-taxonomic-profiling"},{"relation":"model","target_id":"discovery-model-metaphlan"}],"name":"Marker-based microbial profiling","source_ids":["src-discovery-biobakery-metaphlan"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Peak filtering, precursor tolerance, library and candidate set.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["metabolomics"]},"id":"discovery-baseline-mass-spectral-cosine-matching","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym-molecule-retrieval"},{"relation":"model","target_id":"discovery-model-matchms"}],"name":"Mass spectral cosine matching","source_ids":["src-discovery-matchms-matchms"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned motif library, background frequencies and strand convention.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-motif-scanning","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"},{"relation":"model","target_id":"discovery-model-fimo"}],"name":"Motif scanning","source_ids":["src-discovery-meme"],"status":"discovered"} {"attributes":{"applicability":"source_supported","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Task-specific training split and predictor head.","scope_note":"One Hot is an explicitly reported comparator in official TAPE task tables. This record does not imply the same baseline protocol suits every protein task."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-function"]},"id":"discovery-baseline-one-hot-protein-encoding","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-tape"}],"name":"One-hot protein encoding","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned sequence database, MSA construction and score threshold.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-function"]},"id":"discovery-baseline-profile-hmm-sequence-search","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-tape-remote-homology-detection"},{"relation":"model","target_id":"discovery-model-hh-suite"}],"name":"Profile-HMM sequence search","source_ids":["src-discovery-soedinglab-hh-suite"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training reference set and homology leakage controls.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-function"]},"id":"discovery-baseline-protein-homology-transfer","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cafa"},{"relation":"model","target_id":"discovery-model-mmseqs2"}],"name":"Protein homology transfer","source_ids":["src-discovery-soedinglab-mmseqs2"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"learned-specialist","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Pinned backbone, checkpoint and sampling temperature.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["protein-structure"]},"id":"discovery-baseline-protein-sequence-recovery-specialist","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"},{"relation":"model","target_id":"discovery-model-proteinmpnn"}],"name":"Protein sequence recovery specialist","source_ids":["src-discovery-dauparas-proteinmpnn"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Matched candidate edge universe, edge density and seed.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["biological-networks"]},"id":"discovery-baseline-random-regulatory-network","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"}],"name":"Random regulatory network","source_ids":["src-discovery-murali-group-beeline"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"biological-procedure","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Temperature, thermodynamic parameter set and pseudoknot policy.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["rna"]},"id":"discovery-baseline-rna-minimum-free-energy-folding","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beacon"},{"relation":"model","target_id":"discovery-model-viennarna-rnafold"}],"name":"RNA minimum-free-energy folding","source_ids":["src-discovery-viennarna-viennarna"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only GC/codon composition features and held-out split.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["rna"]},"id":"discovery-baseline-rna-sequence-composition-regression","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-mrnabench"}],"name":"RNA sequence-composition regression","source_ids":["src-discovery-morrislab-mrnabench"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Frozen features and donor-disjoint folds; training-only regularization.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["spatial-omics"]},"id":"discovery-baseline-spatial-expression-ridge-regression","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-hest-benchmark"}],"name":"Spatial expression ridge regression","source_ids":["src-discovery-mahmoodlab-hest"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"learned-specialist","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Genome assembly, transcript context and model release.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["genomics"]},"id":"discovery-baseline-specialist-splicing-predictor","kind":"baseline","links":[{"relation":"model","target_id":"discovery-model-spliceai"}],"name":"Specialist splicing predictor","source_ids":["src-discovery-illumina-spliceai"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"simple-statistical","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Training-only expression mean with explicit perturbation averaging.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-training-perturbation-mean","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"Training perturbation mean","source_ids":["src-discovery-altoslabs-perturbench"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"learned-specialist","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Expression normalization, regulator list and training cells.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["biological-networks"]},"id":"discovery-baseline-tree-ensemble-regulatory-inference","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"},{"relation":"model","target_id":"discovery-model-genie3"}],"name":"Tree-ensemble regulatory inference","source_ids":["src-discovery-aertslab-genie3"],"status":"discovered"} {"attributes":{"applicability":"proposed","baseline_type":"null-control","missing_metadata":{"implementation_version":"unextracted","protocol_applicability":"unextracted"},"requirements":"Same cells, preprocessing and metrics as integrated embeddings.","scope_note":"Source documents method or task context; proposed applicability requires protocol review."},"description":"Candidate reference for task-specific evaluation; no score is implied.","facets":{"areas":["single-cell"]},"id":"discovery-baseline-unintegrated-expression-reference","kind":"baseline","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-scib"}],"name":"Unintegrated expression reference","source_ids":["src-discovery-theislab-scib"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Genome reconstruction and taxonomic assignment evaluation","version":null,"profile":{"summary":"AMBER scores metagenomic binning and taxonomic assignments against a supplied gold standard.","sections":[{"title":"Procedure","body":"Provide predicted and reference sequence assignments in the documented binning format. AMBER reports bin-level and sample-level summaries and comparative plots; the reference and sample identifiers must match.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"}],"facts":[{"label":"Record type","value":"Evaluator","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"},{"label":"Inputs","value":"Predicted bin assignments and gold-standard assignments","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"},{"label":"Outputs and assessment","value":"Binning accuracy and completeness summaries","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"}],"strengths":[{"text":"Separates individual-bin quality from sample-wide performance.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"}],"limitations":[{"text":"An evaluator does not define the held-out community, database version or tool fitting procedure.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"}],"diagram":{"title":"Procedure overview","steps":["Sequence assignments","Match reference bins","Compute bin and sample metrics"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-cami-challenge-amber"],"source_locator":"README: introduction; User guide / Input; Metrics computed per bin and per sample"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Genome reconstruction and taxonomic assignment evaluation","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-amber","kind":"benchmark","links":[],"name":"AMBER","source_ids":["src-discovery-cami-challenge-amber"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Three-dimensional molecular learning tasks","version":null,"profile":{"summary":"ATOM3D provides datasets and tooling for learning from three-dimensional molecular structures.","sections":[{"title":"Procedure","body":"Select a molecular task and its dataset, load coordinates and labels, and use the task-specific splitting and evaluation instructions. The package supports structure files and LMDB datasets.","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"}],"facts":[{"label":"Record type","value":"Task suite and data tooling","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"},{"label":"Inputs","value":"Molecular structures and task-specific labels","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"},{"label":"Outputs and assessment","value":"Task-specific molecular predictions","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"}],"strengths":[{"text":"Reusable loading, filtering and splitting utilities support consistent structural-data handling.","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"}],"limitations":[{"text":"The suite name alone does not specify a dataset, split, metric or permitted structural information.","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"}],"diagram":{"title":"Procedure overview","steps":["Select task dataset","Load molecular structures","Fit task predictor","Score task outputs"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-drorlab-atom3d"],"source_locator":"README: Features; Usage / Downloading a dataset; Reference"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Three-dimensional molecular learning tasks","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-atom3d","kind":"benchmark","links":[],"name":"ATOM3D","source_ids":["src-discovery-drorlab-atom3d"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"RNA structure, function and engineering tasks","version":null,"profile":{"summary":"BEACON evaluates RNA representations across structure, function and engineering tasks.","sections":[{"title":"Procedure","body":"Choose the named task directory and matching fine-tuning script. Inputs and prediction heads differ between sequence classification, base-pair maps, degradation, translation and CRISPR-related tasks.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"},{"label":"Inputs","value":"RNA sequences with task-specific labels","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"},{"label":"Outputs and assessment","value":"Classification, regression or structural predictions","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"}],"strengths":[{"text":"Makes multiple RNA task types available through a shared model evaluation codebase.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"}],"limitations":[{"text":"Task scripts and data revisions must be pinned separately; one RNA task score cannot represent all RNA capabilities.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"}],"diagram":{"title":"Procedure overview","steps":["Choose RNA task","Load released split","Fine-tune task head","Evaluate task output"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-terry-r123-rnabenchmark"],"source_locator":"README: Tasks and Datasets; Usage / Finetuning"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"RNA structure, function and engineering tasks","facets":{"areas":["rna"]},"id":"discovery-benchmark-beacon","kind":"benchmark","links":[],"name":"BEACON","source_ids":["src-discovery-terry-r123-rnabenchmark"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Gene regulatory network inference","version":null,"profile":{"summary":"BEELINE evaluates gene regulatory network inference from single-cell expression data.","sections":[{"title":"Procedure","body":"Run a selected inference algorithm, export its ranked regulatory edges and compare those edges with a supplied reference network. The framework separates execution, evaluation and plotting.","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"}],"facts":[{"label":"Record type","value":"Benchmark framework and evaluator","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"},{"label":"Inputs","value":"Single-cell expression and a reference regulatory network","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"},{"label":"Outputs and assessment","value":"Ranked-edge AUPRC, AUROC and early precision","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"}],"strengths":[{"text":"Containerized methods and a common evaluator make methodological comparisons inspectable.","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"}],"limitations":[{"text":"The chosen reference network defines what counts as a correct edge; this is not proof that every inferred interaction is causal.","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"}],"diagram":{"title":"Procedure overview","steps":["Expression data","Infer ranked edges","Compare reference network","Report edge metrics"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-murali-group-beeline"],"source_locator":"README: Usage / BLRunner.py, BLEvaluator.py, BLPlotter.py; Configuration"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Gene regulatory network inference","facets":{"areas":["biological-networks"]},"id":"discovery-benchmark-beeline","kind":"benchmark","links":[],"name":"BEELINE","source_ids":["src-discovery-murali-group-beeline"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"DNA representations on biological downstream tasks","version":null,"profile":{"summary":"BEND evaluates DNA representations on biologically defined downstream tasks.","sections":[{"title":"Procedure","body":"Generate embeddings for the released genomic intervals, then train the supplied supervised predictor or use the relevant unsupervised scoring procedure. Embeddings are expanded to nucleotide resolution according to each tokenizer.","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"},{"label":"Inputs","value":"Genomic intervals, genome sequence and task annotations","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"},{"label":"Outputs and assessment","value":"Gene, regulatory or variant-related task predictions","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"}],"strengths":[{"text":"Includes one-hot and supervised baselines alongside pretrained representations.","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"}],"limitations":[{"text":"Tokenizer upsampling and genomic coordinate handling affect what the predictor receives; task-specific splits and metrics remain necessary.","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"}],"diagram":{"title":"Procedure overview","steps":["Genomic intervals","Compute DNA embeddings","Apply task predictor","Evaluate held-out labels"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-frederikkemarin-bend"],"source_locator":"README: Tutorial / Data format, Computing embeddings, Evaluating models; FAQ / How are embeddings upsampled?"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"DNA representations on biological downstream tasks","facets":{"areas":["genomics"]},"id":"discovery-benchmark-bend","kind":"benchmark","links":[],"name":"BEND","source_ids":["src-discovery-frederikkemarin-bend"],"status":"discovered"} {"attributes":{"entity_level":"challenge","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein function prediction challenge","version":null,"profile":{"summary":"CAFA is a time-based challenge for predicting protein function.","sections":[{"title":"Procedure","body":"Submit ontology-term predictions before the deadline. Proteins receiving experimental annotations after submission become evaluation targets for that challenge round.","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"}],"facts":[{"label":"Record type","value":"Prospective challenge","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"},{"label":"Inputs","value":"Protein sequences and ontology-term predictions","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"},{"label":"Outputs and assessment","value":"Agreement with subsequently acquired functional annotations","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"}],"strengths":[{"text":"Uses later experimental annotations to assess predictions made in advance.","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"}],"limitations":[{"text":"Only proteins that acquire suitable annotations enter the assessed set; ontology version, round and scoring rules must be recorded.","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"}],"diagram":{"title":"Procedure overview","steps":["Release target sequences","Submit functions","Accumulate new annotations","Assess predictions"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-cafa"],"source_locator":"Official CAFA page: The CAFA Challenge, prediction and annotation-growth timeline"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein function prediction challenge","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-cafa","kind":"benchmark","links":[],"name":"CAFA","source_ids":["src-discovery-cafa"],"status":"discovered"} {"attributes":{"entity_level":"challenge","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Metagenomic assembly, binning and profiling assessment","version":null,"profile":{"summary":"CAMI organizes community assessments of metagenomic assembly, binning and taxonomic profiling.","sections":[{"title":"Procedure","body":"Choose a challenge dataset and task, produce its required output and compare against the corresponding reference. Assembly, genome binning, taxonomic binning and abundance profiling are distinct evaluations.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"facts":[{"label":"Record type","value":"Challenge family","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Inputs","value":"Challenge metagenomic data and task-specific submissions","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Outputs and assessment","value":"Separate assembly, binning and profiling assessments","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"strengths":[{"text":"A shared challenge resource supports comparison across the metagenomic analysis workflow.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"limitations":[{"text":"Challenge edition, taxonomy and reference corrections matter; records from different editions are not interchangeable.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"diagram":{"title":"Procedure overview","steps":["Select challenge dataset","Run task method","Submit task output","Compare reference"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Metagenomic assembly, binning and profiling assessment","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami","kind":"benchmark","links":[],"name":"CAMI","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"genome binning","version":null,"profile":{"summary":"Group metagenomic contigs into candidate genomes.","sections":[{"title":"Procedure","body":"This is the CAMI genome binning component, not the parent suite as a whole. Choose a challenge dataset and task, produce its required output and compare against the corresponding reference. Assembly, genome binning, taxonomic binning and abundance profiling are distinct evaluations.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Inputs","value":"Contigs and predicted genome bins","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Outputs and assessment","value":"Genome-bin quality against the reference","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"strengths":[{"text":"A shared challenge resource supports comparison across the metagenomic analysis workflow.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"limitations":[{"text":"Challenge edition, taxonomy and reference corrections matter; records from different editions are not interchangeable.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"diagram":{"title":"Component evaluation overview","steps":["Contigs and predicted genome bins","CAMI genome binning","Genome-bin quality against the reference"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"genome binning","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-genome-binning","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI genome binning","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"metagenome assembly","version":null,"profile":{"summary":"Reconstruct sequences from mixed-community reads.","sections":[{"title":"Procedure","body":"This is the CAMI metagenome assembly component, not the parent suite as a whole. Choose a challenge dataset and task, produce its required output and compare against the corresponding reference. Assembly, genome binning, taxonomic binning and abundance profiling are distinct evaluations.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Inputs","value":"Metagenomic reads and assembled contigs","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Outputs and assessment","value":"Assembly comparison with the challenge reference","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"strengths":[{"text":"A shared challenge resource supports comparison across the metagenomic analysis workflow.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"limitations":[{"text":"Challenge edition, taxonomy and reference corrections matter; records from different editions are not interchangeable.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"diagram":{"title":"Component evaluation overview","steps":["Metagenomic reads and assembled contigs","CAMI metagenome assembly","Assembly comparison with the challenge reference"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"metagenome assembly","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-metagenome-assembly","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI metagenome assembly","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"taxonomic binning","version":null,"profile":{"summary":"Assign individual sequences to taxonomic groups.","sections":[{"title":"Procedure","body":"This is the CAMI taxonomic binning component, not the parent suite as a whole. Choose a challenge dataset and task, produce its required output and compare against the corresponding reference. Assembly, genome binning, taxonomic binning and abundance profiling are distinct evaluations.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Inputs","value":"Reads or contigs with taxonomic assignments","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Outputs and assessment","value":"Assignment quality by taxonomic rank","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"strengths":[{"text":"A shared challenge resource supports comparison across the metagenomic analysis workflow.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"limitations":[{"text":"Challenge edition, taxonomy and reference corrections matter; records from different editions are not interchangeable.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"diagram":{"title":"Component evaluation overview","steps":["Reads or contigs with taxonomic assignments","CAMI taxonomic binning","Assignment quality by taxonomic rank"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"taxonomic binning","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-taxonomic-binning","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI taxonomic binning","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"taxonomic profiling","version":null,"profile":{"summary":"Estimate which taxa occur and their relative abundance.","sections":[{"title":"Procedure","body":"This is the CAMI taxonomic profiling component, not the parent suite as a whole. Choose a challenge dataset and task, produce its required output and compare against the corresponding reference. Assembly, genome binning, taxonomic binning and abundance profiling are distinct evaluations.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Inputs","value":"Sample-level taxon abundance profiles","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},{"label":"Outputs and assessment","value":"Presence and abundance agreement by rank","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"strengths":[{"text":"A shared challenge resource supports comparison across the metagenomic analysis workflow.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"limitations":[{"text":"Challenge edition, taxonomy and reference corrections matter; records from different editions are not interchangeable.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"}],"diagram":{"title":"Component evaluation overview","steps":["Sample-level taxon abundance profiles","CAMI taxonomic profiling","Presence and abundance agreement by rank"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-cami"],"source_locator":"Official CAMI home page: mission; Summary / Per category; CAMI III announcements"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"taxonomic profiling","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-cami-taxonomic-profiling","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-cami"},{"relation":"part_of","target_id":"discovery-benchmark-cami"}],"name":"CAMI taxonomic profiling","source_ids":["src-discovery-cami"],"status":"discovered"} {"attributes":{"entity_level":"challenge","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein interaction docking assessment","version":null,"profile":{"summary":"CAPRI assesses blind predictions of protein-complex structures.","sections":[{"title":"Procedure","body":"Predict the structure of a target complex before its experimental structure is publicly released, then assess submitted models against that structure.","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"}],"facts":[{"label":"Record type","value":"Blind structural challenge","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"},{"label":"Inputs","value":"Released target information for interacting proteins","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"},{"label":"Outputs and assessment","value":"Predicted complex structures evaluated against experiment","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"}],"strengths":[{"text":"Unpublished targets support prospective assessment.","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"}],"limitations":[{"text":"This family record does not pin a target round, allowed input information or scoring implementation.","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"}],"diagram":{"title":"Procedure overview","steps":["Receive complex target","Predict interactions","Reveal reference structure","Assess submitted complexes"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-capri"],"source_locator":"Official CAPRI page: Welcome to the CAPRI web site, first paragraph"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein interaction docking assessment","facets":{"areas":["protein-structure"]},"id":"discovery-benchmark-capri","kind":"benchmark","links":[],"name":"CAPRI","source_ids":["src-discovery-capri"],"status":"discovered"} {"attributes":{"entity_level":"challenge","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Community protein structure assessment","version":null,"profile":{"summary":"CASP assesses protein structure predictions through blind community experiments.","sections":[{"title":"Procedure","body":"Predict the released targets for a particular CASP round and category. Assessors compare submissions with experimental structures using the category's published measures.","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"}],"facts":[{"label":"Record type","value":"Blind structural challenge","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"},{"label":"Inputs","value":"Target sequences and category-specific permitted information","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"},{"label":"Outputs and assessment","value":"Structural agreement with withheld experimental targets","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"}],"strengths":[{"text":"Archived targets, submissions and assessment reports support scrutiny of progress.","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"}],"limitations":[{"text":"Monomer, assembly, refinement and data-assisted tracks involve different inputs and measures; select the exact round and category.","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"}],"diagram":{"title":"Procedure overview","steps":["Select round and track","Submit structural predictions","Release experimental targets","Assess structures"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-casp"],"source_locator":"Prediction Center home page: Welcome; blind prediction; data archive and numerical evaluation sections"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Community protein structure assessment","facets":{"areas":["protein-structure"]},"id":"discovery-benchmark-casp","kind":"benchmark","links":[],"name":"CASP","source_ids":["src-discovery-casp"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Human regulatory DNA representation evaluation","version":null,"profile":{"summary":"DART-Eval examines whether DNA representations capture human gene regulation.","sections":[{"title":"Procedure","body":"Evaluate regulatory-element discrimination, motif footprinting, cell-type specificity, quantitative activity and variant effects using the supplied task resources. Distinguish zero-shot scoring, learned probes and fine-tuned models.","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"},{"label":"Inputs","value":"Human regulatory DNA and task-specific experimental annotations","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"},{"label":"Outputs and assessment","value":"Regulatory task predictions in explicitly different adaptation regimes","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"}],"strengths":[{"text":"Includes increasing task difficulty and ab initio comparators.","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"}],"limitations":[{"text":"Results depend on the adaptation regime and biological task; zero-shot and fine-tuned scores cannot be pooled as the same evaluation.","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"}],"diagram":{"title":"Procedure overview","steps":["Select regulatory task","Choose adaptation regime","Generate predictions","Score task evidence"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-kundajelab-dart-eval"],"source_locator":"README: opening; Tasks 1–5 and their zero-shot, probing, fine-tuning and ab initio subsections"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Human regulatory DNA representation evaluation","facets":{"areas":["genomics"]},"id":"discovery-benchmark-dart-eval","kind":"benchmark","links":[],"name":"DART-Eval","source_ids":["src-discovery-kundajelab-dart-eval"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Generalisation in protein fitness landscapes","version":null,"profile":{"summary":"FLIP tests protein fitness prediction under deliberately different generalization splits.","sections":[{"title":"Procedure","body":"Select an active released split, fit the selected sequence-to-fitness method on its training data and assess its test variants. The repository documents split construction and baseline implementations.","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"}],"facts":[{"label":"Record type","value":"Dataset and split suite","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"},{"label":"Inputs","value":"Protein sequences with experimental fitness measurements","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"},{"label":"Outputs and assessment","value":"Fitness prediction on the selected split","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"}],"strengths":[{"text":"Split definitions make the intended generalization challenge explicit.","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"}],"limitations":[{"text":"Orange splits may overestimate performance; red splits are obsolete and should not support new comparisons.","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"}],"diagram":{"title":"Procedure overview","steps":["Select active split","Fit fitness predictor","Predict held-out variants","Assess fitness predictions"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-j-snackkb-flip"],"source_locator":"README: Folder breakup; Find out more about the splits; Split semaphore"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Generalisation in protein fitness landscapes","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-flip","kind":"benchmark","links":[],"name":"FLIP","source_ids":["src-discovery-j-snackkb-flip"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Expanded protein fitness landscapes","version":null,"profile":{"summary":"FLIP2 extends protein fitness evaluation to additional engineering settings.","sections":[{"title":"Procedure","body":"Use a released split testing mutation count, position, unseen mutations, fitness range or wild-type transfer. Compare zero-shot models, ridge regression and fine-tuned predictors under the same split.","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"}],"facts":[{"label":"Record type","value":"Dataset and split suite","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"},{"label":"Inputs","value":"Protein variant sequences and measured properties","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"},{"label":"Outputs and assessment","value":"Fitness prediction under explicit distribution shifts","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"}],"strengths":[{"text":"Includes simple regression baselines and several practical transfer settings.","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"}],"limitations":[{"text":"Split difficulty changes the question being asked; aggregate rankings do not identify a universal protein engineering method.","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"}],"diagram":{"title":"Procedure overview","steps":["Choose engineering shift","Fit allowed predictor","Predict held-out variants","Compare measured fitness"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-flip2"],"source_locator":"Official FLIP2 site: Abstract; Key Features; Overview; Split Types"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Expanded protein fitness landscapes","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-flip2","kind":"benchmark","links":[],"name":"FLIP2","source_ids":["src-discovery-flip2"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Frozen genomic representations with linear probes","version":null,"profile":{"summary":"GENEB compares frozen DNA representations using a shared linear probe.","sections":[{"title":"Procedure","body":"Encode each DNA sequence, pool its hidden states and fit logistic regression without fine-tuning the encoder. Evaluate released train/test partitions in full-data, ten-shot and one-shot settings using the specified repeated seeds.","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"}],"facts":[{"label":"Record type","value":"Frozen-representation benchmark suite","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"},{"label":"Inputs","value":"DNA classification tasks and frozen sequence embeddings","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"},{"label":"Outputs and assessment","value":"MCC, accuracy and macro-F1 by task and category","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"}],"strengths":[{"text":"A shared probing protocol separates representation quality from model-specific fine-tuning.","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"}],"limitations":[{"text":"This tests frozen embeddings, not the best possible fine-tuned pipeline; full-data and few-shot rankings may differ.","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"}],"diagram":{"title":"Procedure overview","steps":["Released DNA splits","Frozen embeddings","Logistic-regression probe","Task and category metrics"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-darlednik-geneb"],"source_locator":"README: Benchmark; Evaluation protocol; Evaluate your model"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Frozen genomic representations with linear probes","facets":{"areas":["genomics"]},"id":"discovery-benchmark-geneb","kind":"benchmark","links":[],"name":"GENEB","source_ids":["src-discovery-darlednik-geneb"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Genomic sequence classification","version":null,"profile":{"summary":"Genomic Benchmarks packages genomic sequence classification datasets.","sections":[{"title":"Procedure","body":"Download a named and versioned dataset, retain its released training and test folders, then train and assess a classifier. Dataset metadata describe class labels and sequence properties.","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"}],"facts":[{"label":"Record type","value":"Dataset collection and utilities","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"},{"label":"Inputs","value":"Versioned genomic sequence datasets","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"},{"label":"Outputs and assessment","value":"Sequence-classification predictions","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"}],"strengths":[{"text":"Standardized downloads and loaders lower barriers to repeatable data access.","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"}],"limitations":[{"text":"A dataset collection does not enforce one model-fitting or validation protocol; record those choices separately.","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"}],"diagram":{"title":"Procedure overview","steps":["Choose dataset version","Load train and test sequences","Train classifier","Score test predictions"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"source_locator":"README: Usage / info and download_dataset; Structure of package"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Genomic sequence classification","facets":{"areas":["genomics"]},"id":"discovery-benchmark-genomic-benchmarks","kind":"benchmark","links":[],"name":"Genomic Benchmarks","source_ids":["src-discovery-ml-bioinfo-ceitec-genomic-benchmarks"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Glycan properties, taxonomy and molecular interactions","version":null,"profile":{"summary":"GlycanML evaluates glycan learning across taxonomy, immunogenicity, glycosylation and interaction tasks.","sections":[{"title":"Procedure","body":"Represent glycans as sequences or graphs, select a task configuration and train in either a single-task or multi-task setting. Use the corresponding released experiment configuration.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Inputs","value":"Glycan sequences or graphs and task labels","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Outputs and assessment","value":"Task-specific glycan predictions","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"strengths":[{"text":"Supports comparisons between sequence and graph representations.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"limitations":[{"text":"Single-task and multi-task training expose models to different supervision; the configuration must accompany a result.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"diagram":{"title":"Procedure overview","steps":["Select glycan task","Choose sequence or graph representation","Train configured model","Evaluate task"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Glycan properties, taxonomy and molecular interactions","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml","kind":"benchmark","links":[],"name":"GlycanML","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"glycosylation type prediction","version":null,"profile":{"summary":"Predict the glycosylation category associated with a glycan.","sections":[{"title":"Procedure","body":"This is the GlycanML glycosylation type prediction component, not the parent suite as a whole. Represent glycans as sequences or graphs, select a task configuration and train in either a single-task or multi-task setting. Use the corresponding released experiment configuration.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Inputs","value":"Glycan representations","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Outputs and assessment","value":"Glycosylation labels","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"strengths":[{"text":"Supports comparisons between sequence and graph representations.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"limitations":[{"text":"Single-task and multi-task training expose models to different supervision; the configuration must accompany a result.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"diagram":{"title":"Component evaluation overview","steps":["Glycan representations","GlycanML glycosylation type prediction","Glycosylation labels"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"glycosylation type prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-glycosylation-type-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML glycosylation type prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"immunogenicity prediction","version":null,"profile":{"summary":"Predict annotated glycan immunogenicity.","sections":[{"title":"Procedure","body":"This is the GlycanML immunogenicity prediction component, not the parent suite as a whole. Represent glycans as sequences or graphs, select a task configuration and train in either a single-task or multi-task setting. Use the corresponding released experiment configuration.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Inputs","value":"Glycan representations","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Outputs and assessment","value":"Immunogenicity labels","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"strengths":[{"text":"Supports comparisons between sequence and graph representations.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"limitations":[{"text":"Single-task and multi-task training expose models to different supervision; the configuration must accompany a result.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"diagram":{"title":"Component evaluation overview","steps":["Glycan representations","GlycanML immunogenicity prediction","Immunogenicity labels"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"immunogenicity prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-immunogenicity-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML immunogenicity prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"protein-glycan interaction prediction","version":null,"profile":{"summary":"Predict protein–glycan interaction labels.","sections":[{"title":"Procedure","body":"This is the GlycanML protein-glycan interaction prediction component, not the parent suite as a whole. Represent glycans as sequences or graphs, select a task configuration and train in either a single-task or multi-task setting. Use the corresponding released experiment configuration.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Inputs","value":"Protein and glycan information","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Outputs and assessment","value":"Interaction predictions","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"strengths":[{"text":"Supports comparisons between sequence and graph representations.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"limitations":[{"text":"Single-task and multi-task training expose models to different supervision; the configuration must accompany a result.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"diagram":{"title":"Component evaluation overview","steps":["Protein and glycan information","GlycanML protein-glycan interaction prediction","Interaction predictions"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"protein-glycan interaction prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-protein-glycan-interaction-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML protein-glycan interaction prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"taxonomy prediction","version":null,"profile":{"summary":"Predict taxonomy labels associated with glycans.","sections":[{"title":"Procedure","body":"This is the GlycanML taxonomy prediction component, not the parent suite as a whole. Represent glycans as sequences or graphs, select a task configuration and train in either a single-task or multi-task setting. Use the corresponding released experiment configuration.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Inputs","value":"Glycan representations","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},{"label":"Outputs and assessment","value":"Taxonomy labels","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"strengths":[{"text":"Supports comparisons between sequence and graph representations.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"limitations":[{"text":"Single-task and multi-task training expose models to different supervision; the configuration must accompany a result.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"}],"diagram":{"title":"Component evaluation overview","steps":["Glycan representations","GlycanML taxonomy prediction","Taxonomy labels"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-glycanml-glycanml"],"source_locator":"README: Overview; Model Training / Experimental Configurations; Single-Task and Multi-Task Learning"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"taxonomy prediction","facets":{"areas":["glycomics"]},"id":"discovery-benchmark-glycanml-taxonomy-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-glycanml"},{"relation":"part_of","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanML taxonomy prediction","source_ids":["src-discovery-glycanml-glycanml"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Multi-species genome understanding tasks","version":null,"profile":{"summary":"GUE tests genome understanding across multiple datasets, tasks and species.","sections":[{"title":"Procedure","body":"Use the released GUE data and model-specific evaluation scripts. Record the particular dataset, fine-tuning setup and selected checkpoint for each comparison.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"},{"label":"Inputs","value":"Genomic sequences and dataset-specific labels","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"},{"label":"Outputs and assessment","value":"Genome task classification results","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"}],"strengths":[{"text":"Provides evaluation scripts for several genomic model families.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"}],"limitations":[{"text":"The suite-level name does not establish that all model runs used identical checkpoint selection or adaptation.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"}],"diagram":{"title":"Procedure overview","steps":["Choose GUE dataset","Run model-specific fine-tuning","Select checkpoint","Evaluate held-out data"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README: 2.1 GUE: Genome Understanding Evaluation; 6.1 Evaluate models on GUE"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Multi-species genome understanding tasks","facets":{"areas":["genomics"]},"id":"discovery-benchmark-gue","kind":"benchmark","links":[],"name":"GUE","source_ids":["src-discovery-magics-lab-dnabert-2"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Gene expression prediction from matched histology","version":null,"profile":{"summary":"HEST-Benchmark tests gene-expression prediction from matched histology.","sections":[{"title":"Procedure","body":"Encode spatially matched image regions and use the benchmark procedure to predict measured gene expression. This molecular prediction task is distinct from generic medical-image classification.","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"}],"facts":[{"label":"Record type","value":"Spatial transcriptomics benchmark","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"},{"label":"Inputs","value":"Histology regions paired with spatial gene-expression measurements","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"},{"label":"Outputs and assessment","value":"Prediction of selected gene-expression targets","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"}],"strengths":[{"text":"Paired morphology and transcriptomics provide a measurable molecular endpoint.","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"}],"limitations":[{"text":"A correlation with expression does not establish causal regulation; tissue, assay and split definitions remain essential.","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"}],"diagram":{"title":"Procedure overview","steps":["Matched histology regions","Image representation","Expression prediction","Compare spatial measurements"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-mahmoodlab-hest"],"source_locator":"README: What does this repository provide?; HEST-Benchmark; Benchmarking your own model"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Gene expression prediction from matched histology","facets":{"areas":["spatial-omics"]},"id":"discovery-benchmark-hest-benchmark","kind":"benchmark","links":[],"name":"HEST-Benchmark","source_ids":["src-discovery-mahmoodlab-hest"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Molecular identification from tandem mass spectra","version":null,"profile":{"summary":"MassSpecGym evaluates molecular identification and discovery from tandem mass spectra.","sections":[{"title":"Procedure","body":"Choose de novo generation, candidate retrieval or spectrum simulation. Each challenge specifies its input information, splits and output scoring; formula-assisted settings are separate variants.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"facts":[{"label":"Record type","value":"Molecular measurement task suite","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Inputs","value":"MS/MS spectra or molecular structures, depending on task","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Outputs and assessment","value":"Generated molecules, candidate rankings or simulated spectra","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"strengths":[{"text":"Defines complementary tasks around the same measurement modality.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"limitations":[{"text":"Chemical-formula assistance changes available information; tasks and assisted variants cannot be collapsed into one score.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"diagram":{"title":"Procedure overview","steps":["Choose challenge and inputs","Apply released split","Generate task outputs","Score task predictions"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Molecular identification from tandem mass spectra","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym","kind":"benchmark","links":[],"name":"MassSpecGym","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Generate molecular structures from tandem mass spectra","version":null,"profile":{"summary":"Generate candidate molecular structures from a tandem mass spectrum.","sections":[{"title":"Procedure","body":"This is the MassSpecGym De novo molecule generation component, not the parent suite as a whole. Choose de novo generation, candidate retrieval or spectrum simulation. Each challenge specifies its input information, splits and output scoring; formula-assisted settings are separate variants.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Inputs","value":"MS/MS spectrum, optionally a supplied molecular formula","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Outputs and assessment","value":"Generated molecular structures","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"strengths":[{"text":"Defines complementary tasks around the same measurement modality.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"limitations":[{"text":"Chemical-formula assistance changes available information; tasks and assisted variants cannot be collapsed into one score.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"diagram":{"title":"Component evaluation overview","steps":["MS/MS spectrum, optionally a supplied molecular formula","MassSpecGym De novo molecule generation","Generated molecular structures"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Generate molecular structures from tandem mass spectra","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym-de-novo-molecule-generation","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-massspecgym"},{"relation":"part_of","target_id":"discovery-benchmark-massspecgym"}],"name":"MassSpecGym De novo molecule generation","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Rank candidate structures from a tandem mass spectrum","version":null,"profile":{"summary":"Rank candidate molecules for a tandem mass spectrum.","sections":[{"title":"Procedure","body":"This is the MassSpecGym Molecule retrieval component, not the parent suite as a whole. Choose de novo generation, candidate retrieval or spectrum simulation. Each challenge specifies its input information, splits and output scoring; formula-assisted settings are separate variants.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Inputs","value":"MS/MS spectrum and a defined candidate set","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Outputs and assessment","value":"Candidate ranking","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"strengths":[{"text":"Defines complementary tasks around the same measurement modality.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"limitations":[{"text":"Chemical-formula assistance changes available information; tasks and assisted variants cannot be collapsed into one score.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"diagram":{"title":"Component evaluation overview","steps":["MS/MS spectrum and a defined candidate set","MassSpecGym Molecule retrieval","Candidate ranking"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Rank candidate structures from a tandem mass spectrum","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym-molecule-retrieval","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-massspecgym"},{"relation":"part_of","target_id":"discovery-benchmark-massspecgym"}],"name":"MassSpecGym Molecule retrieval","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Predict a tandem mass spectrum from molecular structure","version":null,"profile":{"summary":"Predict a tandem mass spectrum from molecular structure.","sections":[{"title":"Procedure","body":"This is the MassSpecGym Spectrum simulation component, not the parent suite as a whole. Choose de novo generation, candidate retrieval or spectrum simulation. Each challenge specifies its input information, splits and output scoring; formula-assisted settings are separate variants.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Inputs","value":"Molecular structure","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},{"label":"Outputs and assessment","value":"Simulated MS/MS spectrum","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"strengths":[{"text":"Defines complementary tasks around the same measurement modality.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"limitations":[{"text":"Chemical-formula assistance changes available information; tasks and assisted variants cannot be collapsed into one score.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"}],"diagram":{"title":"Component evaluation overview","steps":["Molecular structure","MassSpecGym Spectrum simulation","Simulated MS/MS spectrum"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-pluskal-lab-massspecgym"],"source_locator":"README: opening challenge list; Getting started with MassSpecGym"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Predict a tandem mass spectrum from molecular structure","facets":{"areas":["metabolomics"]},"id":"discovery-benchmark-massspecgym-spectrum-simulation","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-massspecgym"},{"relation":"part_of","target_id":"discovery-benchmark-massspecgym"}],"name":"MassSpecGym Spectrum simulation","source_ids":["src-discovery-pluskal-lab-massspecgym"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"mRNA embedding quality on downstream tasks","version":null,"profile":{"summary":"mRNABench assesses genomic model embeddings on mRNA-specific downstream tasks.","sections":[{"title":"Procedure","body":"Load a named dataset, generate model embeddings and fit a configured linear probe. The documented example uses a homology-aware splitter; retain the splitter and species settings with each evaluation.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"}],"facts":[{"label":"Record type","value":"Representation benchmark suite","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"},{"label":"Inputs","value":"mRNA sequences, task labels and a chosen split","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"},{"label":"Outputs and assessment","value":"Task-specific linear-probe results","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"}],"strengths":[{"text":"Exposes datasets, embedding generation and split selection as explicit steps.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"}],"limitations":[{"text":"The example split is not a universal setting; task data, homology partition and random seed must be recorded.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"}],"diagram":{"title":"Procedure overview","steps":["Choose mRNA dataset","Generate embeddings","Build split and probe","Report held-out metrics"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-morrislab-mrnabench"],"source_locator":"README: opening; Usage; Dataset Catalog"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"mRNA embedding quality on downstream tasks","facets":{"areas":["rna"]},"id":"discovery-benchmark-mrnabench","kind":"benchmark","links":[],"name":"mRNABench","source_ids":["src-discovery-morrislab-mrnabench"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"DNA and RNA fitness prediction","version":null,"profile":{"summary":"NABench evaluates nucleotide models against measured DNA and RNA fitness.","sections":[{"title":"Procedure","body":"Select a nucleic-acid assay and evaluation setting: zero-shot, few-shot, supervised or transfer learning. Compare predictions with the matched assay measurements using that setting's protocol.","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"}],"facts":[{"label":"Record type","value":"Fitness benchmark suite","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"},{"label":"Inputs","value":"DNA or RNA mutant sequences and assay measurements","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"},{"label":"Outputs and assessment","value":"Fitness prediction across distinct adaptation regimes","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"}],"strengths":[{"text":"Collects diverse high-throughput nucleotide assays under a common framework.","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"}],"limitations":[{"text":"Different molecule families and training regimes remain different prediction problems; assay coverage must accompany aggregate scores.","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"}],"diagram":{"title":"Procedure overview","steps":["Select assay and regime","Score nucleotide variants","Compare experimental fitness","Summarize by assay"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-mrzzmrzz-nabench"],"source_locator":"README: Overview; Leaderboard; Baseline Models; Resources"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"DNA and RNA fitness prediction","facets":{"areas":["rna"]},"id":"discovery-benchmark-nabench","kind":"benchmark","links":[],"name":"NABench","source_ids":["src-discovery-mrzzmrzz-nabench"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Metagenomic taxonomic profile evaluation","version":null,"profile":{"summary":"OPAL evaluates predicted microbial taxon abundances.","sections":[{"title":"Procedure","body":"Supply predicted and gold-standard profiles in the documented taxonomic format. Evaluate presence and abundance agreement for matched samples and taxonomic ranks.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"}],"facts":[{"label":"Record type","value":"Evaluator","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"},{"label":"Inputs","value":"Predicted and reference taxonomic abundance profiles","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"},{"label":"Outputs and assessment","value":"Taxonomic profile performance summaries","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"}],"strengths":[{"text":"A shared scorer supports side-by-side profiler assessment.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"}],"limitations":[{"text":"Taxonomy identifiers and ranks must agree with the reference; profiling does not assign each read to a bin.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"}],"diagram":{"title":"Procedure overview","steps":["Taxon abundance profiles","Align ranks and references","Compute profile metrics","Compare profilers"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-cami-challenge-opal"],"source_locator":"README: opening; Computed metrics; Inputs; Running opal.py"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Metagenomic taxonomic profile evaluation","facets":{"areas":["microbiome"]},"id":"discovery-benchmark-opal","kind":"benchmark","links":[],"name":"OPAL","source_ids":["src-discovery-cami-challenge-opal"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Community single-cell analysis benchmarks","version":null,"profile":{"summary":"Open Problems is a community platform for single-cell analysis benchmarks.","sections":[{"title":"Procedure","body":"Choose a specific published task, dataset and evaluation workflow. The platform links benchmark results and datasets; it is not itself one fixed protocol.","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"}],"facts":[{"label":"Record type","value":"Benchmark platform","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"},{"label":"Inputs","value":"Task-specific single-cell data and predictions","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"},{"label":"Outputs and assessment","value":"Task-specific evaluation reports","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"}],"strengths":[{"text":"Provides a common location for community-maintained tasks and results.","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"}],"limitations":[{"text":"The pinned overview is intentionally brief; a concrete task version, metric and split must be extracted before comparing results.","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"}],"diagram":{"title":"Procedure overview","steps":["Select published task","Load task dataset","Run task workflow","Inspect evaluation"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-openproblems-bio-openproblems"],"source_locator":"README: opening and links to benchmarks and datasets; official benchmark directory"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Community single-cell analysis benchmarks","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-open-problems","kind":"benchmark","links":[{"relation":"evaluates_task","target_id":"catalog-task-cell-batch-integration"}],"name":"Open Problems","source_ids":["src-discovery-openproblems-bio-openproblems"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Predicting cellular perturbation responses","version":null,"profile":{"summary":"PerturBench standardizes cellular perturbation prediction workflows.","sections":[{"title":"Procedure","body":"Load a curated single-cell dataset and its defined split, produce predicted perturbed expression, then evaluate an explicitly chosen aggregation and metric pipeline. Some datasets use generated splits; others require released manual partitions.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"}],"facts":[{"label":"Record type","value":"Benchmark framework","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"},{"label":"Inputs","value":"Single-cell expression, perturbation metadata and predicted responses","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"},{"label":"Outputs and assessment","value":"Metrics on aggregated expression or response changes","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"}],"strengths":[{"text":"Separates expression aggregation, distance or correlation metrics and optional ranking assessment.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"}],"limitations":[{"text":"Means, log-fold changes and other representations answer different questions; the exact aggregation and split must accompany the score.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"}],"diagram":{"title":"Procedure overview","steps":["Curated perturbation data","Select released split","Predict response","Aggregate and evaluate"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-altoslabs-perturbench"],"source_locator":"README: Downloading and Preparing Datasets / Data Splitting; Usage / Evaluator Class; Automated Evaluation"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Predicting cellular perturbation responses","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-perturbench","kind":"benchmark","links":[],"name":"PerturBench","source_ids":["src-discovery-altoslabs-perturbench"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Parameter estimation for biological dynamical models","version":null,"profile":{"summary":"The PEtab collection provides data-based parameter-estimation problems for mechanistic models.","sections":[{"title":"Procedure","body":"Choose a model problem together with its experimental measurements and parameter definitions. Fit the mathematical model using a specified solver and optimization configuration.","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"}],"facts":[{"label":"Record type","value":"Mechanistic model problem collection","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"},{"label":"Inputs","value":"Mathematical models, observations, conditions and parameters","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"},{"label":"Outputs and assessment","value":"Parameter fits and problem-specific optimization assessments","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"}],"strengths":[{"text":"Packages experimental data with the models to be fitted.","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"}],"limitations":[{"text":"A problem definition does not standardize solver tolerances, starting points, optimization budgets or a universal score.","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"}],"diagram":{"title":"Procedure overview","steps":["Select PEtab problem","Configure solver and parameters","Fit experimental observations","Assess fit and computation"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"source_locator":"README: introduction; Overview problem table"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Parameter estimation for biological dynamical models","facets":{"areas":["mechanistic-biology"]},"id":"discovery-benchmark-petab-benchmark-collection","kind":"benchmark","links":[],"name":"PEtab benchmark collection","source_ids":["src-discovery-benchmarking-initiative-benchmark-models-petab"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein representation evaluation","version":null,"profile":{"summary":"PFMBench provides protein foundation model evaluation across downstream tasks.","sections":[{"title":"Procedure","body":"Select a task, dataset, model and tuning configuration in its Hydra-based workflow. Keep fine-tuned tasks separate from zero-shot scoring procedures.","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"},{"label":"Inputs","value":"Protein representations and task-specific data","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"},{"label":"Outputs and assessment","value":"Structure, function and other protein task metrics","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"}],"strengths":[{"text":"A modular workflow makes models and tuning configurations explicit.","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"}],"limitations":[{"text":"Task-specific data releases and metric settings require separate pinning; the task count is not a comparable scientific score.","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"}],"diagram":{"title":"Procedure overview","steps":["Select task configuration","Load model and data","Train or score","Evaluate task"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-biomap-research-pfmbench"],"source_locator":"README: Overview; Features; Project Structure; Quick Start"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein representation evaluation","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-pfmbench","kind":"benchmark","links":[],"name":"PFMBench","source_ids":["src-discovery-biomap-research-pfmbench"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein-ligand interaction evaluation","version":null,"profile":{"summary":"PLINDER combines protein–ligand interaction data with evaluation resources.","sections":[{"title":"Procedure","body":"Pin the dataset release and iteration, choose a train/validation/test partition and evaluate a docking method on the selected held-out subset. Test subsets distinguish ligand, pocket and protein novelty.","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"}],"facts":[{"label":"Record type","value":"Dataset and evaluation resource","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"},{"label":"Inputs","value":"Protein–ligand structures and annotated splits","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"},{"label":"Outputs and assessment","value":"Docking evaluation on specified novelty strata","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"}],"strengths":[{"text":"Similarity annotations and stratified test sets help make generalization explicit.","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"}],"limitations":[{"text":"Dataset release-date corrections and split revisions can change membership; the two-part data version must be retained.","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"}],"diagram":{"title":"Procedure overview","steps":["Pin data release","Choose novelty split","Predict complexes","Evaluate selected subset"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-plinder-org-plinder"],"source_locator":"README: About; Plinder versions; Gold standard benchmark sets; Known bugs"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein-ligand interaction evaluation","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-plinder","kind":"benchmark","links":[],"name":"PLINDER","source_ids":["src-discovery-plinder-org-plinder"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Geometric and chemical plausibility of molecular poses","version":null,"profile":{"summary":"PoseBusters checks geometric and chemical plausibility of molecular poses.","sections":[{"title":"Procedure","body":"Provide a generated ligand pose, optionally with a receptor and reference ligand, and run the appropriate molecule, docking or redocking checks. These check types require different inputs.","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"}],"facts":[{"label":"Record type","value":"Evaluator","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"},{"label":"Inputs","value":"Ligand coordinates; receptor and reference ligand where required","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"},{"label":"Outputs and assessment","value":"Validity checks appropriate to the supplied configuration","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"}],"strengths":[{"text":"Adds molecular plausibility checks to positional assessment.","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"}],"limitations":[{"text":"Passing geometry checks does not demonstrate binding affinity, biological activity or correct ranking of compounds.","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"}],"diagram":{"title":"Procedure overview","steps":["Load molecular pose","Choose check configuration","Run validity checks","Inspect failures"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-maabuu-posebusters"],"source_locator":"README: Usage, CLI and PoseBusters configuration examples"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Geometric and chemical plausibility of molecular poses","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-posebusters","kind":"benchmark","links":[],"name":"PoseBusters","source_ids":["src-discovery-maabuu-posebusters"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein prediction, design and dynamics evaluation","version":null,"profile":{"summary":"ProteinBench evaluates protein prediction, design and dynamics across several dimensions.","sections":[{"title":"Procedure","body":"Choose the relevant modality-to-modality task and examine its quality, novelty, diversity and robustness measures. Structure-conditioned design and backbone generation use different evaluation panels.","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"}],"facts":[{"label":"Record type","value":"Evaluation framework","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"},{"label":"Inputs","value":"Protein sequences, structures or task-specific design conditions","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"},{"label":"Outputs and assessment","value":"Separate quality, novelty, diversity and robustness measures","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"}],"strengths":[{"text":"Makes trade-offs visible beyond a single generation-quality score.","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"}],"limitations":[{"text":"Novelty and diversity need to be interpreted alongside quality; a computed design metric is not experimental validation.","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"}],"diagram":{"title":"Procedure overview","steps":["Choose protein task","Generate predictions or designs","Measure quality and diversity","Inspect trade-offs"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-proteinbench"],"source_locator":"Official ProteinBench site: Abstract; inverse-folding and backbone-design table captions"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein prediction, design and dynamics evaluation","facets":{"areas":["protein-structure"]},"id":"discovery-benchmark-proteinbench","kind":"benchmark","links":[],"name":"ProteinBench","source_ids":["src-discovery-proteinbench"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein variant effect prediction","version":null,"profile":{"summary":"ProteinGym evaluates mutation-effect predictors against experimental protein assays.","sections":[{"title":"Procedure","body":"Score each variant in a pinned assay release, evaluate within assays and then apply the published protein and function-category aggregation. Keep substitutions, indels, zero-shot and supervised tracks separate.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"}],"facts":[{"label":"Record type","value":"Assay and evaluation suite","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"},{"label":"Inputs","value":"Protein variants, assay measurements and permitted model inputs","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"},{"label":"Outputs and assessment","value":"Spearman and other track-specific metrics","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"}],"strengths":[{"text":"Provides assay-level results and aggregation intended to reduce repeated-protein bias.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"}],"limitations":[{"text":"DMS fitness is assay-specific; MSA- and structure-informed methods use different inputs from sequence-only methods.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"}],"diagram":{"title":"Procedure overview","steps":["Pin assay release","Score protein variants","Compute assay metrics","Aggregate by protocol"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-oatml-markslab-proteingym"],"source_locator":"README: Overview; Results; Resources; Usage and reproducibility"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein variant effect prediction","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-proteingym","kind":"benchmark","links":[{"relation":"evaluates_task","target_id":"catalog-task-proteingym-effects"}],"name":"ProteinGym","source_ids":["src-discovery-oatml-markslab-proteingym"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Single-cell integration evaluation","version":null,"profile":{"summary":"scIB measures both batch removal and biological preservation in single-cell integration.","sections":[{"title":"Procedure","body":"Preprocess annotated single-cell data, run an integration method and evaluate its representation. Inspect biological-conservation metrics alongside batch-correction metrics; the package, reusable pipeline and published study are separate resources.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"}],"facts":[{"label":"Record type","value":"Evaluator and associated benchmarking study","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"},{"label":"Inputs","value":"Single-cell expression or chromatin data with batch and biological annotations","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"},{"label":"Outputs and assessment","value":"Biological-conservation and batch-correction metric panels","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"}],"strengths":[{"text":"Makes the two competing integration objectives visible.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"}],"limitations":[{"text":"A well-mixed embedding can erase real biological differences; label-dependent scores also depend on the reference annotations.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"}],"diagram":{"title":"Procedure overview","steps":["Annotated cell data","Integrate batches","Measure biological conservation","Measure batch correction"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-theislab-scib"],"source_locator":"README: Package scib; Metrics; Integration Tools; Resources"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Single-cell integration evaluation","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-scib","kind":"benchmark","links":[{"relation":"evaluates_task","target_id":"catalog-task-cell-batch-integration"}],"name":"scIB","source_ids":["src-discovery-theislab-scib"],"status":"discovered"} {"attributes":{"entity_level":"evaluator","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Perturbation prediction metric calibration","version":null,"profile":{"summary":"scPertEval examines evaluation protocols for single-cell perturbation predictions.","sections":[{"title":"Procedure","body":"Specify the representation, metric, score transformation and reporting strategy. Use the software to score predictions or calibrate a protocol against its built-in controls.","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"}],"facts":[{"label":"Record type","value":"Evaluation software and protocol calibration","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"},{"label":"Inputs","value":"Predicted and observed perturbation responses","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"},{"label":"Outputs and assessment","value":"Protocol-dependent scores and control calibration","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"}],"strengths":[{"text":"Treats the choice of evaluation protocol as something to test explicitly.","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"}],"limitations":[{"text":"A score is not interpretable without its representation and transformation; the calibration controls are not new biological measurements.","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"}],"diagram":{"title":"Procedure overview","steps":["Choose protocol components","Score predictions","Evaluate controls","Report calibrated interpretation"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"source_locator":"README: introduction; Quick start"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Perturbation prediction metric calibration","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-scperteval","kind":"benchmark","links":[],"name":"scPertEval","source_ids":["src-discovery-virtual-cell-research-community-scperteval"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Protein representation learning tasks","version":null,"profile":{"summary":"TAPE evaluates protein embeddings on five supervised downstream tasks.","sections":[{"title":"Procedure","body":"Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"}],"facts":[{"label":"Record type","value":"Task suite","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"},{"label":"Inputs","value":"Protein sequences and task-specific labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"},{"label":"Outputs and assessment","value":"Classification, contact precision or rank-correlation metrics","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"}],"diagram":{"title":"Procedure overview","steps":["Select downstream task","Fit task predictor","Evaluate held-out proteins","Report task metric"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: introduction; List of Models and Tasks; Data; Leaderboard"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Protein representation learning tasks","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape","kind":"benchmark","links":[],"name":"TAPE","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Contact Prediction","version":null,"profile":{"summary":"Predict residue contacts from protein sequence.","sections":[{"title":"Procedure","body":"This is the TAPE Contact Prediction component, not the parent suite as a whole. Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"},{"label":"Inputs","value":"ProteinNet sequence and structural-contact labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"},{"label":"Outputs and assessment","value":"Precision among the top L/5 medium- and long-range contacts","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"}],"diagram":{"title":"Component evaluation overview","steps":["ProteinNet sequence and structural-contact labels","TAPE Contact Prediction","Precision among the top L/5 medium- and long-range contacts"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Contact Prediction"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Contact Prediction","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-contact-prediction","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Contact Prediction","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Fluorescence","version":null,"profile":{"summary":"Predict measured fluorescence from protein sequence.","sections":[{"title":"Procedure","body":"This is the TAPE Fluorescence component, not the parent suite as a whole. Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"},{"label":"Inputs","value":"Protein variants and fluorescence labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"},{"label":"Outputs and assessment","value":"Spearman rank correlation","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"}],"diagram":{"title":"Component evaluation overview","steps":["Protein variants and fluorescence labels","TAPE Fluorescence","Spearman rank correlation"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Fluorescence"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Fluorescence","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-fluorescence","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Fluorescence","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Remote Homology Detection","version":null,"profile":{"summary":"Classify remote protein homology.","sections":[{"title":"Procedure","body":"This is the TAPE Remote Homology Detection component, not the parent suite as a whole. Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"},{"label":"Inputs","value":"Protein sequences and homology labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"},{"label":"Outputs and assessment","value":"Top-1 classification accuracy","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"}],"diagram":{"title":"Component evaluation overview","steps":["Protein sequences and homology labels","TAPE Remote Homology Detection","Top-1 classification accuracy"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Remote Homology Detection"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Remote Homology Detection","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-remote-homology-detection","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Remote Homology Detection","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Secondary Structure","version":null,"profile":{"summary":"Predict secondary-structure labels for protein residues.","sections":[{"title":"Procedure","body":"This is the TAPE Secondary Structure component, not the parent suite as a whole. Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"},{"label":"Inputs","value":"Protein sequences and residue labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"},{"label":"Outputs and assessment","value":"Three-class accuracy","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"}],"diagram":{"title":"Component evaluation overview","steps":["Protein sequences and residue labels","TAPE Secondary Structure","Three-class accuracy"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Secondary Structure"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Secondary Structure","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-secondary-structure","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Secondary Structure","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"task","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Stability","version":null,"profile":{"summary":"Predict measured protein stability from sequence.","sections":[{"title":"Procedure","body":"This is the TAPE Stability component, not the parent suite as a whole. Choose secondary structure, contact prediction, remote homology, fluorescence or stability. Train a task head and evaluate using the corresponding data and metric. The current PyTorch repository differs from the original publication code.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"}],"facts":[{"label":"Record type","value":"Component task","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"},{"label":"Inputs","value":"Protein sequences and stability labels","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"},{"label":"Outputs and assessment","value":"Spearman rank correlation","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"}],"strengths":[{"text":"Includes simple one-hot comparators alongside pretrained representations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"}],"limitations":[{"text":"The maintainers explicitly advise using the original implementation to reproduce the original paper.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"}],"diagram":{"title":"Component evaluation overview","steps":["Protein sequences and stability labels","TAPE Stability","Spearman rank correlation"],"caption":"Conceptual component-task overview. Exact data versions, split files and scorer settings must be taken from the source.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README: List of Models and Tasks; Data; Leaderboard / Stability"},"coverage":"limited","gaps":["Component-specific split manifest and complete scoring configuration have not yet been extracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Existing catalogue evidence and cited extraction inspected. The specific protocol gaps listed here remain unresolved; numerical source checks do not constitute complete methods review."}}},"description":"Stability","facets":{"areas":["protein-function"]},"id":"discovery-benchmark-tape-stability","kind":"benchmark","links":[{"relation":"parent","target_id":"discovery-benchmark-tape"},{"relation":"part_of","target_id":"discovery-benchmark-tape"}],"name":"TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"suite","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Molecular binding, biochemical activity and related specialist tasks","version":null,"profile":{"summary":"TDC organizes datasets and evaluation tools for therapeutic machine learning.","sections":[{"title":"Procedure","body":"For rewire, select an in-scope molecular binding, activity or molecular-design task, then specify the dataset, split and evaluator. The broader TDC platform also includes tasks outside this catalogue's molecular remit.","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"}],"facts":[{"label":"Record type","value":"Task platform, molecular subset","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"},{"label":"Inputs","value":"Task-specific molecular or biochemical data","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"},{"label":"Outputs and assessment","value":"Dataset-specific predictions and evaluation","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"}],"strengths":[{"text":"Provides dataset loaders, split utilities and evaluation tools.","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"}],"limitations":[{"text":"The full platform includes clinical tasks outside rewire's scope; inclusion must be decided per task, not by platform name.","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"}],"diagram":{"title":"Procedure overview","steps":["Select molecular task","Pin dataset and split","Predict biochemical outcome","Evaluate chosen metric"],"caption":"Conceptual overview of the cited procedure; consult the pinned source for executable settings.","source_ids":["src-discovery-mims-harvard-tdc"],"source_locator":"README: Unique Features; Design of TDC; Data Loaders"},"coverage":"reviewed","gaps":["Exact dataset release, split manifest and evaluator configuration must be attached to each numerical evaluation."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is a profile review, not an independent execution or numerical reproduction."}}},"description":"Molecular binding, biochemical activity and related specialist tasks","facets":{"areas":["molecular-interactions"]},"id":"discovery-benchmark-tdc-molecular-tasks","kind":"benchmark","links":[],"name":"TDC molecular tasks","source_ids":["src-discovery-mims-harvard-tdc"],"status":"discovered"} {"attributes":{"entity_level":"challenge","missing_metadata":{"dataset_release":"unextracted","metric_implementation":"unextracted","split_manifest":"unextracted","version":"unextracted"},"scope_note":"Specialist molecular or omics evaluation; protocol details require review before numerical comparison.","task":"Zero-shot perturbation prediction in unseen cellular contexts","version":null,"profile":{"summary":"The 2026 Virtual Cell Challenge tests perturbation prediction in unseen cellular contexts.","sections":[{"title":"Procedure","body":"Predict post-CRISPRi expression from non-targeting-control cells and target-gene identifiers. No challenge-specific training responses are released. Three cell lines support validation and three are reserved for final testing.","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"}],"facts":[{"label":"Record type","value":"Prospective challenge","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"},{"label":"Inputs","value":"Unperturbed expression and CRISPRi target genes","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"},{"label":"Assessment","value":"Held-out expression responses, scored with the challenge metric panel","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"}],"strengths":[{"text":"Tests transfer between cellular contexts rather than interpolation among measured responses in one context.","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"}],"limitations":[{"text":"External training data are allowed; their provenance must be recorded. The final 2026 assessment is not yet available at this review date.","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"}],"diagram":{"title":"Procedure overview","steps":["Unperturbed cells and targets","Predict CRISPRi response","Withheld experimental response","Apply challenge scorer"],"caption":"Conceptual overview, not an executable specification.","source_ids":["src-discovery-vcc2026"],"source_locator":"Arc announcement, 20 August 2026: The 2026 task; Evaluation"},"coverage":"reviewed","gaps":["Pin the final cell-eval version, metric definitions and aggregation rules before interpreting final scores."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}},"description":"Zero-shot perturbation prediction in unseen cellular contexts","facets":{"areas":["single-cell"]},"id":"discovery-benchmark-virtual-cell-challenge-2026","kind":"benchmark","links":[],"name":"Virtual Cell Challenge 2026","source_ids":["src-discovery-vcc2026"],"status":"discovered"} {"attributes":{"assay":"mass spectrometry","missing_metadata":{"split":"not_applicable","version":"unextracted"},"scope_note":"Reference library; a leakage-aware benchmark split and scoring protocol must be defined separately.","split":null,"version":null},"description":"Experimental lipid reference spectra for identification assessment.","facets":{"areas":["lipidomics"]},"id":"discovery-dataset-lipid-maps-standards-spectra","kind":"dataset","links":[],"name":"LIPID MAPS Standards Spectra","source_ids":["src-discovery-lipidmaps-spectra"],"status":"discovered"} {"attributes":{"missing_metadata":{"denominator":"unextracted","split":"unextracted","version":"unreported"},"split":null,"version":null},"description":"Dataset used by the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-dataset-tape-fluorescence-source-dataset","kind":"dataset","links":[],"name":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"missing_metadata":{"denominator":"unextracted","split":"unextracted","version":"unreported"},"split":null,"version":null},"description":"Dataset used by the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-dataset-tape-stability-source-dataset","kind":"dataset","links":[],"name":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-bepler-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-bepler"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence Bepler leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-lstm-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-lstm"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence LSTM leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-one-hot-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-one-hot"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence One Hot leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-resnet-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-resnet"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence ResNet leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-transformer-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-transformer"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence Transformer leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-fluorescence","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Fluorescence","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-fluorescence-unirep-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-unirep"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-fluorescence"},{"relation":"dataset","target_id":"discovery-dataset-tape-fluorescence-source-dataset"}],"name":"TAPE Fluorescence Unirep leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-bepler-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-bepler"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability Bepler leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-lstm-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-lstm"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability LSTM leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-one-hot-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-one-hot"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability One Hot leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-resnet-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-resnet"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability ResNet leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-transformer-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-transformer"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability Transformer leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"comparison":{"adaptation":null,"aggregation":null,"budget":null,"dataset_version":null,"inputs":null,"metric_implementation":null,"population":null,"protocol_id":"discovery-benchmark-tape-stability","split":null},"missing_metadata":{"checkpoint":"unreported","denominator":"unextracted","seeds":"unreported","split_manifest":"unextracted"},"origin":"author_reported","protocol":"TAPE Stability","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Historical benchmark-maintainer report; experiments have not been rerun.","facets":{"areas":["protein-function"]},"id":"discovery-evaluation-tape-stability-unirep-leaderboard-evaluation","kind":"evaluation","links":[{"relation":"model","target_id":"discovery-model-tape-unirep"},{"relation":"benchmark","target_id":"discovery-benchmark-tape-stability"},{"relation":"dataset","target_id":"discovery-dataset-tape-stability-source-dataset"}],"name":"TAPE Stability Unirep leaderboard evaluation","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-geneb"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Agro Nucleotide Transformer","version":null,"profile":{"summary":"Agro Nucleotide Transformer: plant genomic representation family","sections":[{"title":"Available evidence","body":"The discovery record links to instadeepai/nucleotide-transformer official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'Agro Nucleotide Transformer' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Plant genomic representation family","facets":{"areas":["genomics"]},"id":"discovery-model-agro-nucleotide-transformer","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-geneb"}],"name":"Agro Nucleotide Transformer","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-plinder"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"AlphaFold 3","version":null,"profile":{"summary":"AlphaFold 3: biomolecular complex structure prediction","sections":[{"title":"Available evidence","body":"The discovery record links to google-deepmind/alphafold3 official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-google-deepmind-alphafold3"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-google-deepmind-alphafold3"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'AlphaFold 3' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Biomolecular complex structure prediction","facets":{"areas":["protein-structure"]},"id":"discovery-model-alphafold-3","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-plinder"}],"name":"AlphaFold 3","source_ids":["src-discovery-google-deepmind-alphafold3"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-posebusters"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"AutoDock Vina","version":null,"profile":{"summary":"AutoDock Vina is a conventional docking engine for searching ligand conformations against a receptor.","sections":[{"title":"How it works","body":"A scoring function guides gradient-based conformational search. Candidate poses are ranked under the selected docking configuration; receptor preparation and search settings are part of the method.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"facts":[{"label":"Method class","value":"Docking search with a scoring function","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"strengths":[{"text":"Provides a procedural comparator with batch and multiple-ligand workflows.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"limitations":[{"text":"A docking score is not an experimentally measured affinity. Pose and affinity tasks require separate evaluation.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"}],"diagram":{"title":"Conceptual procedure","steps":["Prepared receptor and ligand","Search configuration","Conformation search","Scoring function","Ranked docking poses"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"source_locator":"README.md: AutoDock Vina description and Features"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Procedural docking and virtual screening","facets":{"areas":["molecular-interactions"]},"id":"discovery-model-autodock-vina","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-posebusters"}],"name":"AutoDock Vina","source_ids":["src-discovery-ccsb-scripps-autodock-vina"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Basenji","version":null,"profile":{"summary":"Basenji predicts regulatory activity along DNA with deep convolutional networks.","sections":[{"title":"How it works","body":"Sequence inputs are processed by convolutional models that output activity in bins. Variant scoring compares predicted activity for alternate sequences in a specified genomic context.","source_ids":["src-discovery-calico-basenji"],"source_locator":"README.md: opening goals and Changes versus Basset"}],"facts":[{"label":"Output form","value":"Regulatory activity in sequence bins","source_ids":["src-discovery-calico-basenji"],"source_locator":"README.md: opening goals and Changes versus Basset"}],"strengths":[{"text":"Supports sequence-to-activity and variant-scoring workflows.","source_ids":["src-discovery-calico-basenji"],"source_locator":"README.md: opening goals and Changes versus Basset"}],"limitations":[{"text":"Bin resolution, target tracks and sequence context affect what the prediction represents; a project-level record is not a specific trained regulatory model.","source_ids":["src-discovery-calico-basenji"],"source_locator":"README.md: opening goals and Changes versus Basset"}],"diagram":{"title":"Conceptual procedure","steps":["DNA sequence","Convolutional model","Binned regulatory activity","Allele comparison","Predicted regulatory effect"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-calico-basenji"],"source_locator":"README.md: opening goals and Changes versus Basset"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Sequence-to-regulatory-profile prediction","facets":{"areas":["genomics"]},"id":"discovery-model-basenji","kind":"model","links":[],"name":"Basenji","source_ids":["src-discovery-calico-basenji"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-plinder"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Boltz","version":null,"profile":{"summary":"Boltz is a biomolecular interaction model family. Boltz-2 adds affinity prediction to complex-structure prediction.","sections":[{"title":"Boltz-2 architecture","body":"The Boltz-2 implementation combines molecular and alignment features with a Pairformer module. A conditioned diffusion module predicts coordinates; a separate affinity module produces binding outputs. This architecture description applies to Boltz-2, not automatically to every Boltz family release.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"src/boltz/model/models/boltz2.py at the pinned repository revision: MSAModule, PairformerModule, DiffusionConditioning and AffinityModule; README Inference"},{"title":"How it works","body":"The documented YAML input describes the biomolecules and requested properties. Structure prediction and affinity outputs are distinct: one affinity output estimates binding strength, while another classifies binders against decoys.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"facts":[{"label":"Access","value":"Repository states code and models use the MIT licence","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"strengths":[{"text":"Supports structure and affinity workflows in one openly distributed project.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"limitations":[{"text":"Binder probability and affinity regression are trained with different supervision and must not be compared as the same metric. Unqualified CLI calls select the latest model.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"README.md: Introduction, Inference, Understanding the affinity prediction and License"}],"diagram":{"title":"Conceptual procedure","steps":["Molecular inputs","MSA / Pairformer features","Coordinate diffusion","Structure","Separate affinity module"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-jwohlwend-boltz"],"source_locator":"Pinned repository src/boltz/model/models/boltz2.py, module construction and forward; README affinity prediction"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Biomolecular structure and affinity model family","facets":{"areas":["molecular-interactions"]},"id":"discovery-model-boltz","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-plinder"}],"name":"Boltz","source_ids":["src-discovery-jwohlwend-boltz"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cami"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"CAMISIM","version":null,"profile":{"summary":"CAMISIM simulates shotgun metagenome datasets from modelled microbial-community abundances.","sections":[{"title":"How it works","body":"Community composition and sequencing configuration are used to generate synthetic metagenomic reads. The simulator can support benchmark construction but is not itself a taxonomic classifier.","source_ids":["src-discovery-cami-challenge-camisim"],"source_locator":"README.md: introduction and update to version 2.0"}],"facts":[{"label":"Record role","value":"Data-generation procedure","source_ids":["src-discovery-cami-challenge-camisim"],"source_locator":"README.md: introduction and update to version 2.0"}],"strengths":[{"text":"Produces controlled synthetic data for evaluating metagenomic workflows.","source_ids":["src-discovery-cami-challenge-camisim"],"source_locator":"README.md: introduction and update to version 2.0"}],"limitations":[{"text":"Simulation assumptions constrain realism. The documented CAMISIM2 and legacy 1.31-final workflows must be treated as distinct versions.","source_ids":["src-discovery-cami-challenge-camisim"],"source_locator":"README.md: introduction and update to version 2.0"}],"diagram":{"title":"Conceptual procedure","steps":["Community configuration","Abundance model","Read simulation","Synthetic metagenome","Benchmark input"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-cami-challenge-camisim"],"source_locator":"README.md: introduction and update to version 2.0"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Microbial community and metagenome simulation","facets":{"areas":["microbiome"]},"id":"discovery-model-camisim","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami"}],"name":"CAMISIM","source_ids":["src-discovery-cami-challenge-camisim"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-dart-eval"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ChromBPNet","version":null,"profile":{"summary":"ChromBPNet predicts chromatin-accessibility patterns at base resolution while separating assay bias from regulatory sequence signal.","sections":[{"title":"How it works","body":"The documented workflow trains bias-factorised sequence models for accessibility measurements. Predictions can be used for sequence interpretation and variant analysis, with assay and preprocessing settings retained.","source_ids":["src-discovery-kundajelab-chrombpnet"],"source_locator":"README.md: title, paper description and tutorial links"}],"facts":[{"label":"Output","value":"Base-resolution chromatin-accessibility predictions","source_ids":["src-discovery-kundajelab-chrombpnet"],"source_locator":"README.md: title, paper description and tutorial links"}],"strengths":[{"text":"Explicitly models assay bias when learning accessibility patterns.","source_ids":["src-discovery-kundajelab-chrombpnet"],"source_locator":"README.md: title, paper description and tutorial links"}],"limitations":[{"text":"A trained model is tied to its assay data and preprocessing; candidate applications are not verified benchmark results.","source_ids":["src-discovery-kundajelab-chrombpnet"],"source_locator":"README.md: title, paper description and tutorial links"}],"diagram":{"title":"Conceptual procedure","steps":["DNA and accessibility data","Bias-factorised model training","Sequence prediction","Base-resolution accessibility","Interpretation or variant analysis"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-kundajelab-chrombpnet"],"source_locator":"README.md: title, paper description and tutorial links"},"coverage":"reviewed","gaps":["This narrative covers the documented purpose; layer counts and checkpoint-specific training settings remain unextracted."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Bias-aware chromatin accessibility prediction","facets":{"areas":["genomics"]},"id":"discovery-model-chrombpnet","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"}],"name":"ChromBPNet","source_ids":["src-discovery-kundajelab-chrombpnet"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"COBRApy","version":null,"profile":{"summary":"COBRApy provides constraint-based analyses of metabolic networks, rather than a pretrained neural checkpoint.","sections":[{"title":"How it works","body":"A metabolic reconstruction and constraints are passed to an optimisation solver. Analyses include flux balance, flux variability and gene-deletion experiments. Results depend on the reconstruction, objective, bounds and solver.","source_ids":["src-discovery-opencobra-cobrapy"],"source_locator":"README.rst: What is COBRApy?"}],"facts":[{"label":"Method class","value":"Constraint-based metabolic modelling","source_ids":["src-discovery-opencobra-cobrapy"],"source_locator":"README.rst: What is COBRApy?"}],"strengths":[{"text":"Provides explicit modelling assumptions and several mechanistic analysis procedures.","source_ids":["src-discovery-opencobra-cobrapy"],"source_locator":"README.rst: What is COBRApy?"}],"limitations":[{"text":"Software identity alone is insufficient to reproduce an analysis; biological constraints and solver settings must be recorded.","source_ids":["src-discovery-opencobra-cobrapy"],"source_locator":"README.rst: What is COBRApy?"}],"diagram":{"title":"Conceptual procedure","steps":["Metabolic reconstruction","Bounds and objective","Optimisation solver","Flux or deletion analysis","Predicted metabolic behaviour"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-opencobra-cobrapy"],"source_locator":"README.rst: What is COBRApy?"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Constraint-based metabolic modelling","facets":{"areas":["mechanistic-biology"]},"id":"discovery-model-cobrapy","kind":"model","links":[],"name":"COBRApy","source_ids":["src-discovery-opencobra-cobrapy"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-casp"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ColabFold","version":null,"profile":{"summary":"ColabFold: protein folding pipeline","sections":[{"title":"Available evidence","body":"The discovery record links to sokrypton/ColabFold official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-sokrypton-colabfold"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-sokrypton-colabfold"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ColabFold' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Protein folding pipeline","facets":{"areas":["protein-structure"]},"id":"discovery-model-colabfold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-casp"}],"name":"ColabFold","source_ids":["src-discovery-sokrypton-colabfold"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-gue"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"DNABERT-2","version":null,"profile":{"summary":"DNABERT-2 is a DNA encoder pretrained on sequences from multiple species. It supplies representations that can be adapted to genomic tasks.","sections":[{"title":"How it works","body":"DNA is compressed into variable-length byte-pair tokens. A BERT-style encoder uses ALiBi positional biases; masked-language pretraining learns contextual features. Sequence pooling or a separately trained prediction head produces task outputs.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"facts":[{"label":"Released model","value":"DNABERT-2-117M","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"},{"label":"Objective","value":"Masked-language pretraining","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"strengths":[{"text":"One released encoder can support embedding extraction and supervised adaptation across several genomic tasks.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"limitations":[{"text":"An encoder embedding is not a splice-impact prediction. Pooling, sequence context and the supervised head are part of the evaluated pipeline.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"}],"diagram":{"title":"Conceptual procedure","steps":["DNA sequence","Byte-pair tokens","BERT encoder with ALiBi","Token or pooled embeddings","Task-specific head"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-magics-lab-dnabert-2"],"source_locator":"README.md: Introduction, Model and Data, Quick Start, Pre-Training and Fine-tune"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Genomic sequence representation model","facets":{"areas":["genomics"]},"id":"discovery-model-dnabert-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-gue"}],"name":"DNABERT-2","source_ids":["src-discovery-magics-lab-dnabert-2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-massspecgym"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"DreaMS","version":null,"profile":{"summary":"DreaMS learns molecular representations from tandem mass spectra for downstream interpretation tasks.","sections":[{"title":"How it works","body":"A transformer is pretrained on unannotated spectra using masked spectral peaks and chromatographic retention ordering. Representations feed task-specific prediction or similarity workflows.","source_ids":["src-discovery-pluskal-lab-dreams"],"source_locator":"README.md: introduction and What can I do with DreaMS?"}],"facts":[{"label":"Training resource","value":"GeMS unannotated MS/MS spectra","source_ids":["src-discovery-pluskal-lab-dreams"],"source_locator":"README.md: introduction and What can I do with DreaMS?"}],"strengths":[{"text":"The project provides representations, spectra resources and downstream analysis workflows.","source_ids":["src-discovery-pluskal-lab-dreams"],"source_locator":"README.md: introduction and What can I do with DreaMS?"}],"limitations":[{"text":"An embedding or similarity score is not a definitive chemical identification; the candidate set and evaluation split remain essential.","source_ids":["src-discovery-pluskal-lab-dreams"],"source_locator":"README.md: introduction and What can I do with DreaMS?"}],"diagram":{"title":"Conceptual procedure","steps":["MS/MS spectrum","Spectral preprocessing","DreaMS transformer","Spectrum embedding","Task head or similarity"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-pluskal-lab-dreams"],"source_locator":"README.md: introduction and What can I do with DreaMS?"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Tandem mass spectrum representation model","facets":{"areas":["metabolomics"]},"id":"discovery-model-dreams","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym"}],"name":"DreaMS","source_ids":["src-discovery-pluskal-lab-dreams"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteingym"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESM-1v","version":null,"profile":{"summary":"ESM-1v: protein variant effect model family","sections":[{"title":"Available evidence","body":"The discovery record links to facebookresearch/esm official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ESM-1v' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Protein variant effect model family","facets":{"areas":["protein-function"]},"id":"discovery-model-esm-1v","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteingym"}],"name":"ESM-1v","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteingym"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESM-2","version":null,"profile":{"summary":"ESM-2 is a family of protein sequence transformers that produce residue-level and sequence-level representations.","sections":[{"title":"How it works","body":"Amino-acid tokens pass through a pretrained transformer. Hidden states can be retained for each residue or pooled for a whole protein; downstream tasks need an explicit scoring rule or predictor. ESMFold adds a structure-prediction system and is a separate pipeline.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"facts":[{"label":"Training resource","value":"UniRef-derived protein sequences","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"},{"label":"Configuration distinction","value":"The catalogue 8M entry is not the 650M or 15B checkpoint","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"strengths":[{"text":"Embeddings can be extracted directly from individual sequences; the repository provides several model sizes.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"limitations":[{"text":"Different parameter sizes, pooling methods and supervised heads are not interchangeable evaluations.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"}],"diagram":{"title":"Conceptual procedure","steps":["Protein sequence","Amino-acid tokens","ESM-2 transformer","Residue embeddings","Pooling or task predictor"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: Main models, Getting started, Compute embeddings and Pre-trained Models"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Protein sequence representation family","facets":{"areas":["protein-function"]},"id":"discovery-model-esm-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteingym"}],"name":"ESM-2","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESM-IF1","version":null,"profile":{"summary":"ESM-IF1: protein inverse folding model","sections":[{"title":"Available evidence","body":"The discovery record links to facebookresearch/esm official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ESM-IF1' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Protein inverse folding model","facets":{"areas":["protein-structure"]},"id":"discovery-model-esm-if1","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ESM-IF1","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESM3","version":null,"profile":{"summary":"ESM3: multimodal protein model family","sections":[{"title":"Available evidence","body":"The discovery record links to evolutionaryscale/esm official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ESM3' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Multimodal protein model family","facets":{"areas":["protein-structure"]},"id":"discovery-model-esm3","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ESM3","source_ids":["src-discovery-evolutionaryscale-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESMC","version":null,"profile":{"summary":"ESMC: protein representation family","sections":[{"title":"Available evidence","body":"The discovery record links to evolutionaryscale/esm official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ESMC' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Protein representation family","facets":{"areas":["protein-function"]},"id":"discovery-model-esmc","kind":"model","links":[],"name":"ESMC","source_ids":["src-discovery-evolutionaryscale-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESMFold","version":null,"profile":{"summary":"ESMFold predicts protein structures from individual amino-acid sequences using an ESM-2 representation model and a folding system.","sections":[{"title":"How it works","body":"The sequence is embedded and converted into a three-dimensional structure. The implementation exposes recycling and chunking controls; those choices affect memory use and the exact evaluated run.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"facts":[{"label":"Primary output","value":"PDB structure with confidence information","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"strengths":[{"text":"The documented inference interface produces a structure directly from sequence.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"limitations":[{"text":"Long sequences and larger batches can exceed device memory. Version v0 and v1 refer to different released models.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"}],"diagram":{"title":"Conceptual procedure","steps":["Protein sequence","ESM-2 features","Folding system","Recycling","Predicted structure"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-facebookresearch-esm"],"source_locator":"README.md: ESMFold Structure Prediction and Main models"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Protein structure prediction model","facets":{"areas":["protein-structure"]},"id":"discovery-model-esmfold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ESMFold","source_ids":["src-discovery-facebookresearch-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ESMFold2","version":null,"profile":{"summary":"ESMFold2: protein structure and complex prediction","sections":[{"title":"Available evidence","body":"The discovery record links to evolutionaryscale/esm official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-evolutionaryscale-esm"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'ESMFold2' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Protein structure and complex prediction","facets":{"areas":["protein-structure"]},"id":"discovery-model-esmfold2","kind":"model","links":[],"name":"ESMFold2","source_ids":["src-discovery-evolutionaryscale-esm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-geneb"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Evo 2","version":null,"profile":{"summary":"Evo 2 models and generates DNA at nucleotide resolution using the StripedHyena 2 architecture.","sections":[{"title":"How it works","body":"An autoregressive sequence model predicts successive nucleotides from preceding context. Scoring and generation use the selected released checkpoint; context length and device requirements depend on that configuration.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"facts":[{"label":"Training resource","value":"OpenGenome2","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"},{"label":"Objective","value":"Autoregressive sequence prediction","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"strengths":[{"text":"The family is designed for long-context sequence modelling, with released inference code.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"limitations":[{"text":"A family-level maximum context length does not establish the settings used by a particular published evaluation.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"}],"diagram":{"title":"Conceptual procedure","steps":["DNA nucleotides","StripedHyena 2","Autoregressive predictions","Sequence scoring or generation"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-arcinstitute-evo2"],"source_locator":"README.md: introduction, Models and Installation"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Genome sequence modelling family","facets":{"areas":["genomics"]},"id":"discovery-model-evo-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-geneb"}],"name":"Evo 2","source_ids":["src-discovery-arcinstitute-evo2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-dart-eval"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"FIMO","version":null,"profile":{"summary":"FIMO searches sequences for matches to supplied motifs. It is a procedural motif-scanning comparator.","sections":[{"title":"How it works","body":"A motif set and sequence collection are supplied with an alphabet and background model. The background accounts for letter-frequency differences when scoring motif occurrences.","source_ids":["src-discovery-meme"],"source_locator":"FIMO submission documentation: Input motifs, Input sequences and Background model"}],"facts":[{"label":"Inputs","value":"Motifs, sequences and background model","source_ids":["src-discovery-meme"],"source_locator":"FIMO submission documentation: Input motifs, Input sequences and Background model"}],"strengths":[{"text":"Makes motif identity and background assumptions explicit.","source_ids":["src-discovery-meme"],"source_locator":"FIMO submission documentation: Input motifs, Input sequences and Background model"}],"limitations":[{"text":"Changing the background or motif collection changes the analysis. Motif occurrence is not by itself proof of binding or regulatory activity.","source_ids":["src-discovery-meme"],"source_locator":"FIMO submission documentation: Input motifs, Input sequences and Background model"}],"diagram":{"title":"Conceptual procedure","steps":["Motifs and sequences","Alphabet and background","Motif scanning","Candidate motif occurrences"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-meme"],"source_locator":"FIMO submission documentation: Input motifs, Input sequences and Background model"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Sequence motif scanning","facets":{"areas":["genomics"]},"id":"discovery-model-fimo","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-dart-eval"}],"name":"FIMO","source_ids":["src-discovery-meme"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-perturbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"GEARS","version":null,"profile":{"summary":"GEARS predicts transcriptional responses to genetic perturbations using single-cell perturbation-screen data.","sections":[{"title":"How it works","body":"A task-specific model is trained on measured perturbations, then predicts gene-expression responses for requested single or combined perturbations. Training composition determines what generalisation question is being tested.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"facts":[{"label":"Required evidence","value":"Perturbation identities and cells per condition","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"strengths":[{"text":"The implementation explicitly supports single-gene and multi-gene perturbation workflows.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"limitations":[{"text":"The maintainers state that cross-cell-type transfer is unsupported and that reliable combinatorial prediction needs some combinatorial training data.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"}],"diagram":{"title":"Conceptual procedure","steps":["Perturbation-screen cells","Training perturbations","GEARS predictor","Requested perturbation","Expression response"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-snap-stanford-gears"],"source_locator":"README.md: introduction and Important notes"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Genetic perturbation response prediction","facets":{"areas":["single-cell"]},"id":"discovery-model-gears","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"GEARS","source_ids":["src-discovery-snap-stanford-gears"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Genie 3","version":null,"profile":{"summary":"Genie 3: equivariant all-atom protein design","sections":[{"title":"Available evidence","body":"The discovery record links to aqlaboratory/genie3 official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-aqlaboratory-genie3"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-aqlaboratory-genie3"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'Genie 3' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Equivariant all-atom protein design","facets":{"areas":["protein-structure"]},"id":"discovery-model-genie-3","kind":"model","links":[],"name":"Genie 3","source_ids":["src-discovery-aqlaboratory-genie3"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beeline"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"GENIE3","version":null,"profile":{"summary":"GENIE3 infers candidate gene regulatory networks from gene-expression data using ensembles of trees.","sections":[{"title":"How it works","body":"Gene expression supplies the measurements used by the tree-ensemble inference method. Its output is a candidate network whose biological interpretation requires a specified evaluation protocol.","source_ids":["src-discovery-aertslab-genie3"],"source_locator":"README.md: GENIE3 introduction"}],"facts":[{"label":"Identity distinction","value":"GENIE3 network inference is separate from Genie 3 protein design","source_ids":["src-discovery-aertslab-genie3"],"source_locator":"README.md: GENIE3 introduction"}],"strengths":[{"text":"Provides an established non-foundation-model comparator for network inference.","source_ids":["src-discovery-aertslab-genie3"],"source_locator":"README.md: GENIE3 introduction"}],"limitations":[{"text":"Network prediction requires external reference edges or interventions for validation; the project name does not define a matched benchmark.","source_ids":["src-discovery-aertslab-genie3"],"source_locator":"README.md: GENIE3 introduction"}],"diagram":{"title":"Conceptual procedure","steps":["Gene-expression data","Tree-ensemble inference","Regulator relevance","Candidate network"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-aertslab-genie3"],"source_locator":"README.md: GENIE3 introduction"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Tree-ensemble gene regulatory network inference","facets":{"areas":["biological-networks"]},"id":"discovery-model-genie3","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"}],"name":"GENIE3","source_ids":["src-discovery-aertslab-genie3"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-glycanml"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"GlycanGT","version":null,"profile":{"summary":"GlycanGT learns glycan representations with a graph transformer that treats both sugars and linkages as tokens.","sections":[{"title":"How it works","body":"Node and edge tokens include content, identifiers and token types. Transformer layers produce a graph-level embedding. Masked pretraining also supports prediction of missing glycan components.","source_ids":["src-discovery-matsui-lab-glycangt"],"source_locator":"README.md: Model architecture / Training details"}],"facts":[{"label":"Architecture","value":"TokenGT-based graph transformer","source_ids":["src-discovery-matsui-lab-glycangt"],"source_locator":"README.md: Model architecture / Training details"}],"strengths":[{"text":"Explicit linkage tokens allow the representation to include more than monosaccharide composition.","source_ids":["src-discovery-matsui-lab-glycangt"],"source_locator":"README.md: Model architecture / Training details"}],"limitations":[{"text":"The training collection excludes ambiguous symbols; completion predictions are hypotheses about missing structure, not experimental confirmation.","source_ids":["src-discovery-matsui-lab-glycangt"],"source_locator":"README.md: Model architecture / Training details"}],"diagram":{"title":"Conceptual procedure","steps":["Glycan graph","Sugar and linkage tokens","Graph transformer","Graph embedding","Classification or completion"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-matsui-lab-glycangt"],"source_locator":"README.md: Model architecture / Training details"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Glycan graph transformer","facets":{"areas":["glycomics"]},"id":"discovery-model-glycangt","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-glycanml"}],"name":"GlycanGT","source_ids":["src-discovery-matsui-lab-glycangt"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beeline"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"GRNBoost","version":null,"profile":{"summary":"GRNBoost infers candidate regulatory networks by predicting gene expression with boosted trees.","sections":[{"title":"How it works","body":"For each target gene, expression from candidate regulators is used in a regression. Feature importance becomes evidence for candidate regulatory edges; the implementation distributes these regressions with Spark.","source_ids":["src-discovery-aertslab-grnboost"],"source_locator":"README.md: What is GRNBoost? and algorithm description"}],"facts":[{"label":"Implementation distinction","value":"This source describes GRNBoost; GRNBoost2 is not silently substituted","source_ids":["src-discovery-aertslab-grnboost"],"source_locator":"README.md: What is GRNBoost? and algorithm description"}],"strengths":[{"text":"Offers a scalable conventional learning comparator for network inference.","source_ids":["src-discovery-aertslab-grnboost"],"source_locator":"README.md: What is GRNBoost? and algorithm description"}],"limitations":[{"text":"Predictive importance in observational expression data is not proof of a direct causal regulatory edge.","source_ids":["src-discovery-aertslab-grnboost"],"source_locator":"README.md: What is GRNBoost? and algorithm description"}],"diagram":{"title":"Conceptual procedure","steps":["Expression matrix","Candidate regulators","Per-gene boosted regressions","Feature importance","Candidate regulatory network"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-aertslab-grnboost"],"source_locator":"README.md: What is GRNBoost? and algorithm description"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Boosted-tree regulatory network inference","facets":{"areas":["biological-networks"]},"id":"discovery-model-grnboost","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beeline"}],"name":"GRNBoost","source_ids":["src-discovery-aertslab-grnboost"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cafa"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"HH-suite","version":null,"profile":{"summary":"HH-suite searches for protein homologues by comparing sequence profiles represented as hidden Markov models.","sections":[{"title":"How it works","body":"A query profile is compared with profiles in a reference database. The resulting alignments and similarity evidence support homology-based analysis.","source_ids":["src-discovery-soedinglab-hh-suite"],"source_locator":"README.md: introduction"}],"facts":[{"label":"Method class","value":"Hidden Markov model profile alignment","source_ids":["src-discovery-soedinglab-hh-suite"],"source_locator":"README.md: introduction"}],"strengths":[{"text":"Provides a non-neural homology comparator using evolutionary profile information.","source_ids":["src-discovery-soedinglab-hh-suite"],"source_locator":"README.md: introduction"}],"limitations":[{"text":"Its input information includes a profile and database; it is not directly comparable to a single-sequence model without accounting for that extra information.","source_ids":["src-discovery-soedinglab-hh-suite"],"source_locator":"README.md: introduction"}],"diagram":{"title":"Conceptual procedure","steps":["Protein query profile","Reference profile database","HMM–HMM alignment","Ranked homologues"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-soedinglab-hh-suite"],"source_locator":"README.md: introduction"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Profile hidden Markov sequence search","facets":{"areas":["protein-function"]},"id":"discovery-model-hh-suite","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cafa"}],"name":"HH-suite","source_ids":["src-discovery-soedinglab-hh-suite"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"HUMAnN","version":null,"profile":{"summary":"HUMAnN: microbial functional profiling","sections":[{"title":"Available evidence","body":"The discovery record links to biobakery/humann official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-biobakery-humann"],"source_locator":"readme.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-biobakery-humann"],"source_locator":"readme.md"}],"coverage":"limited","gaps":["The linked readme.md has not yielded a reviewed architecture or procedure for the configuration 'HUMAnN' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Microbial functional profiling","facets":{"areas":["microbiome"]},"id":"discovery-model-humann","kind":"model","links":[],"name":"HUMAnN","source_ids":["src-discovery-biobakery-humann"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cami"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Kraken 2","version":null,"profile":{"summary":"Kraken 2: sequence-based metagenomic classification","sections":[{"title":"Available evidence","body":"The discovery record links to DerrickWood/kraken2 official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-derrickwood-kraken2"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-derrickwood-kraken2"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'Kraken 2' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Sequence-based metagenomic classification","facets":{"areas":["microbiome"]},"id":"discovery-model-kraken-2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami"}],"name":"Kraken 2","source_ids":["src-discovery-derrickwood-kraken2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"LipidBlast","version":null,"profile":{"summary":"LipidBlast is an in-silico tandem mass spectral library used for lipid annotation by library search.","sections":[{"title":"How it works","body":"Experimentally acquired MS/MS spectra are compared with a computer-generated reference library. The library and search procedure together determine the candidate annotations.","source_ids":["src-discovery-lipidblast"],"source_locator":"Project page: LipidBlast in silico tandem mass spectrometry database for lipid identification; FAQ"}],"facts":[{"label":"Method class","value":"Generated spectral library, not a learned checkpoint","source_ids":["src-discovery-lipidblast"],"source_locator":"Project page: LipidBlast in silico tandem mass spectrometry database for lipid identification; FAQ"}],"strengths":[{"text":"Provides a procedural reference spanning multiple lipid classes and instrument types.","source_ids":["src-discovery-lipidblast"],"source_locator":"Project page: LipidBlast in silico tandem mass spectrometry database for lipid identification; FAQ"}],"limitations":[{"text":"Library matching depends on fragment information and acquisition conditions; a matching candidate does not resolve every structural isomer.","source_ids":["src-discovery-lipidblast"],"source_locator":"Project page: LipidBlast in silico tandem mass spectrometry database for lipid identification; FAQ"}],"diagram":{"title":"Conceptual procedure","steps":["Lipid MS/MS spectrum","Library search settings","In-silico reference spectra","Spectral matches","Candidate lipid annotations"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-lipidblast"],"source_locator":"Project page: LipidBlast in silico tandem mass spectrometry database for lipid identification; FAQ"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Rule-based lipid fragmentation library matching","facets":{"areas":["lipidomics"]},"id":"discovery-model-lipidblast","kind":"model","links":[],"name":"LipidBlast","source_ids":["src-discovery-lipidblast"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"LipidFinder","version":null,"profile":{"summary":"LipidFinder: lC-MS lipid feature filtering and annotation","sections":[{"title":"Available evidence","body":"The discovery record links to lipidfinder official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-lipidfinder"],"source_locator":"HTML main page"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-lipidfinder"],"source_locator":"HTML main page"}],"coverage":"limited","gaps":["The linked HTML main page has not yielded a reviewed architecture or procedure for the configuration 'LipidFinder' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","The linked project page returned HTTP 403 during this profile review; no architecture claims were extracted from that response."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"LC-MS lipid feature filtering and annotation","facets":{"areas":["lipidomics"]},"id":"discovery-model-lipidfinder","kind":"model","links":[],"name":"LipidFinder","source_ids":["src-discovery-lipidfinder"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-massspecgym"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"matchms","version":null,"profile":{"summary":"matchms is a toolkit for cleaning, processing and comparing tandem mass spectra.","sections":[{"title":"How it works","body":"Spectra and metadata are imported, filtered and validated before a selected pairwise similarity measure is applied. Learned similarity plug-ins are separate methods from the core processing toolkit.","source_ids":["src-discovery-matchms-matchms"],"source_locator":"README.rst: opening description and supported workflows"}],"facts":[{"label":"Method class","value":"Mass-spectral processing and comparison software","source_ids":["src-discovery-matchms-matchms"],"source_locator":"README.rst: opening description and supported workflows"}],"strengths":[{"text":"Makes preprocessing and similarity choices explicit and extensible.","source_ids":["src-discovery-matchms-matchms"],"source_locator":"README.rst: opening description and supported workflows"}],"limitations":[{"text":"Different filtering rules or similarity plug-ins define different pipelines; the toolkit name is not a unique evaluated model.","source_ids":["src-discovery-matchms-matchms"],"source_locator":"README.rst: opening description and supported workflows"}],"diagram":{"title":"Conceptual procedure","steps":["MS/MS files","Metadata and peak cleaning","Similarity measure","Pairwise comparison","Scores and matches"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-matchms-matchms"],"source_locator":"README.rst: opening description and supported workflows"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Mass spectral processing and similarity matching","facets":{"areas":["metabolomics"]},"id":"discovery-model-matchms","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym"}],"name":"matchms","source_ids":["src-discovery-matchms-matchms"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cami"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"MetaPhlAn","version":null,"profile":{"summary":"MetaPhlAn: marker-based microbial profiling","sections":[{"title":"Available evidence","body":"The discovery record links to biobakery/MetaPhlAn official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-biobakery-metaphlan"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-biobakery-metaphlan"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'MetaPhlAn' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Marker-based microbial profiling","facets":{"areas":["microbiome"]},"id":"discovery-model-metaphlan","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cami"}],"name":"MetaPhlAn","source_ids":["src-discovery-biobakery-metaphlan"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-cafa"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"MMseqs2","version":null,"profile":{"summary":"MMseqs2 is a toolkit for searching and clustering large protein and nucleotide sequence collections.","sections":[{"title":"How it works","body":"A query collection is searched against a specified sequence or profile database, or clustered using configured similarity criteria. The selected command and database define the operational method.","source_ids":["src-discovery-soedinglab-mmseqs2"],"source_locator":"README.md: introduction and supported workflows"}],"facts":[{"label":"Method class","value":"Sequence search and clustering software","source_ids":["src-discovery-soedinglab-mmseqs2"],"source_locator":"README.md: introduction and supported workflows"}],"strengths":[{"text":"Provides established sequence-similarity procedures that can act as task-specific comparators.","source_ids":["src-discovery-soedinglab-mmseqs2"],"source_locator":"README.md: introduction and supported workflows"}],"limitations":[{"text":"Database contents, coverage thresholds and sensitivity settings must be matched before interpreting comparisons with learned models.","source_ids":["src-discovery-soedinglab-mmseqs2"],"source_locator":"README.md: introduction and supported workflows"}],"diagram":{"title":"Conceptual procedure","steps":["Query sequences","Reference database","Search or clustering","Alignments or clusters"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-soedinglab-mmseqs2"],"source_locator":"README.md: introduction and supported workflows"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Sequence search and clustering","facets":{"areas":["protein-function"]},"id":"discovery-model-mmseqs2","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-cafa"}],"name":"MMseqs2","source_ids":["src-discovery-soedinglab-mmseqs2"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-massspecgym"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"MSAlign","version":null,"profile":{"summary":"MSAlign retrieves candidate molecules from tandem mass spectra using aligned representations.","sections":[{"title":"How it works","body":"Frozen DreaMS spectrum and ChemBERTa molecule encoders feed lightweight projections. Candidate-based contrastive training aligns their representations; retrieval ranks molecules from the specified candidate set.","source_ids":["src-discovery-msalign"],"source_locator":"arXiv:2605.19752v1 abstract"}],"facts":[{"label":"Evaluated task","value":"Molecule retrieval from MS/MS","source_ids":["src-discovery-msalign"],"source_locator":"arXiv:2605.19752v1 abstract"}],"strengths":[{"text":"Reuses pretrained encoders while training smaller alignment components.","source_ids":["src-discovery-msalign"],"source_locator":"arXiv:2605.19752v1 abstract"}],"limitations":[{"text":"Retrieval difficulty depends on candidate construction and splitting. The paper explicitly examines the tradeoff between leakage and distribution shift.","source_ids":["src-discovery-msalign"],"source_locator":"arXiv:2605.19752v1 abstract"}],"diagram":{"title":"Conceptual procedure","steps":["Spectrum and candidate molecules","Frozen encoders","Projection networks","Shared representation","Candidate ranking"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-msalign"],"source_locator":"arXiv:2605.19752v1 abstract"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Spectrum-to-molecule representation alignment","facets":{"areas":["metabolomics"]},"id":"discovery-model-msalign","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-massspecgym"}],"name":"MSAlign","source_ids":["src-discovery-msalign"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-geneb"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Nucleotide Transformer","version":null,"profile":{"summary":"Nucleotide Transformer: genomic representation model family","sections":[{"title":"Available evidence","body":"The discovery record links to instadeepai/nucleotide-transformer official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'Nucleotide Transformer' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Genomic representation model family","facets":{"areas":["genomics"]},"id":"discovery-model-nucleotide-transformer","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-geneb"}],"name":"Nucleotide Transformer","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-casp"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"OpenFold","version":null,"profile":{"summary":"OpenFold: trainable protein structure prediction implementation","sections":[{"title":"Available evidence","body":"The discovery record links to aqlaboratory/openfold official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-aqlaboratory-openfold"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-aqlaboratory-openfold"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'OpenFold' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Trainable protein structure prediction implementation","facets":{"areas":["protein-structure"]},"id":"discovery-model-openfold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-casp"}],"name":"OpenFold","source_ids":["src-discovery-aqlaboratory-openfold"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"Pangolin","version":null,"profile":{"summary":"Pangolin predicts changes in splice-site strength from DNA variants. It accepts variant files or custom sequence inputs.","sections":[{"title":"Architecture","body":"Pangolin uses 16 residual blocks with dilated convolutions and skip connections. Separate outputs estimate splice-site probability and usage across heart, liver, brain and testis. The published model was trained using sequence and splicing measurements from human, rhesus macaque, rat and mouse.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"Original paper linked in README: https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture; Results: Pangolin predicts splice site usage"},{"title":"How it works","body":"Reference genome and transcript annotation define the sequence context. The neural predictor estimates splice-site strength; the command-line tool reports the largest positive and negative changes near each variant. Masking optionally removes particular gains and losses at annotated sites.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"facts":[{"label":"Masking","value":"Default mask=True; a mask=False evaluation is a distinct configuration","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"strengths":[{"text":"Provides changes in splice strength and their positions, with configurable scoring distance and masking.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"limitations":[{"text":"Only substitutions and simple indels are supported by the documented interface. Missing gene annotations, reference mismatches and chromosome-edge cases can exclude variants.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"README.md: introduction and Usage (command-line), including supported variants and --mask"}],"diagram":{"title":"Conceptual procedure","steps":["DNA context","Dilated residual convolutions","Tissue-specific outputs","Splice strength","Variant-induced change"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-tkzeng-pangolin"],"source_locator":"Original paper linked in README: https://doi.org/10.1186/s13059-022-02664-4, Figure 1 and Methods: Deep neural network architecture; README Usage"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Splice site strength prediction","facets":{"areas":["genomics"]},"id":"discovery-model-pangolin","kind":"model","links":[],"name":"Pangolin","source_ids":["src-discovery-tkzeng-pangolin"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ProteinMPNN","version":null,"profile":{"summary":"ProteinMPNN designs amino-acid sequences for a supplied protein backbone, with controls for fixed residues and chains.","sections":[{"title":"How it works","body":"A parsed structure and design constraints are supplied to the sequence-design model. It samples amino-acid sequences conditional on the backbone; sampling temperature changes diversity. Full-backbone and Cα-only weights are separate configurations.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"facts":[{"label":"Catalogue weight name","value":"v_48_020","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"},{"label":"Output","value":"Designed sequences and model scores","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"strengths":[{"text":"Allows selected chains and positions to be redesigned while retaining specified residues.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"limitations":[{"text":"Requires a suitable input structure. Sequence generation does not itself demonstrate folding, activity or experimental success.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"}],"diagram":{"title":"Conceptual procedure","steps":["Backbone structure","Chain / residue constraints","ProteinMPNN","Conditional sequence sampling","Designed sequences"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-dauparas-proteinmpnn"],"source_locator":"README.md: Full protein backbone models, CA only models and protein_mpnn_run.py arguments"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Structure-conditioned protein sequence design","facets":{"areas":["protein-structure"]},"id":"discovery-model-proteinmpnn","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"ProteinMPNN","source_ids":["src-discovery-dauparas-proteinmpnn"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-proteinbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"RFdiffusion","version":null,"profile":{"summary":"RFdiffusion generates protein structures with optional conditioning such as motifs or target information.","sections":[{"title":"How it works","body":"A diffusion-based structure-generation workflow samples designs subject to the selected conditioning. Sequence design and experimental testing are subsequent steps rather than guaranteed properties of the generated backbone.","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"source_locator":"README.md: What is RFdiffusion? and supported design challenges"}],"facts":[{"label":"Output","value":"Generated protein structures","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"source_locator":"README.md: What is RFdiffusion? and supported design challenges"}],"strengths":[{"text":"Supports unconditional generation and constrained protein-design problems.","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"source_locator":"README.md: What is RFdiffusion? and supported design challenges"}],"limitations":[{"text":"A generated structure is a candidate design, not evidence of an experimentally functional binder.","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"source_locator":"README.md: What is RFdiffusion? and supported design challenges"}],"diagram":{"title":"Conceptual procedure","steps":["Design constraints","Structure diffusion","Generated backbone","Separate sequence design","Experimental evaluation"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"source_locator":"README.md: What is RFdiffusion? and supported design challenges"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Protein structure generation","facets":{"areas":["protein-structure"]},"id":"discovery-model-rfdiffusion","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-proteinbench"}],"name":"RFdiffusion","source_ids":["src-discovery-rosettacommons-rfdiffusion"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beacon"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"RNA-FM","version":null,"profile":{"summary":"RNA-FM is a pretrained RNA sequence encoder for structural and functional representation learning.","sections":[{"title":"How it works","body":"A BERT-style transformer encodes RNA tokens into contextual embeddings after self-supervised sequence training. Structural or functional predictions require the corresponding downstream model; RNA-FM alone should not be labelled as a complete 3D folding pipeline.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"facts":[{"label":"RNA-FM training","value":"Repository reports more than 23 million non-coding RNA sequences","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"strengths":[{"text":"Reusable representations do not require experimental labels during pretraining.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"limitations":[{"text":"The ncRNA encoder and the coding-sequence mRNA-FM extension have different training modalities and should not share checkpoint identities.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"}],"diagram":{"title":"Conceptual procedure","steps":["RNA sequence","RNA tokens","Pretrained transformer","Contextual embeddings","Task-specific predictor"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-ml4bio-rna-fm"],"source_locator":"README.md: introduction and Foundation Models and Extended Ecosystem"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"RNA sequence representation family","facets":{"areas":["rna"]},"id":"discovery-model-rna-fm","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beacon"}],"name":"RNA-FM","source_ids":["src-discovery-ml4bio-rna-fm"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-perturbench"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"scGPT","version":null,"profile":{"summary":"scGPT learns representations of genes and cells from single-cell measurements. Its pretrained checkpoints support task-specific adaptation.","sections":[{"title":"How it works","body":"Gene identifiers and expression values are encoded together and processed by a transformer. The resulting representations support cell embeddings or task heads. Vocabulary, preprocessing and the selected checkpoint must accompany any result.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"facts":[{"label":"Whole-human training","value":"Repository reports 33 million normal human cells","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"strengths":[{"text":"The repository provides whole-human and specialised checkpoints, plus workflows for annotation, integration and perturbation tasks.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"limitations":[{"text":"Whole-human, organ-specific and continually pretrained checkpoints are different configurations. A pretraining claim does not establish transfer performance in a new cell population.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"}],"diagram":{"title":"Conceptual procedure","steps":["Gene IDs and expression","Gene / value encoders","Transformer","Cell and gene representations","Adapted task output"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-bowang-lab-scgpt"],"source_locator":"README.md: Pretrained scGPT checkpoints and Tutorials; scgpt/model/model.py at cebd6fae655b9c585a4807daa3ac31bb764f06b4, lines 79–188"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Single-cell multi-omics model","facets":{"areas":["single-cell"]},"id":"discovery-model-scgpt","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-perturbench"}],"name":"scGPT","source_ids":["src-discovery-bowang-lab-scgpt"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-scib"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"scVI","version":null,"profile":{"summary":"scVI: probabilistic single-cell expression model","sections":[{"title":"Available evidence","body":"The discovery record links to scverse/scvi-tools official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-scverse-scvi-tools"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-scverse-scvi-tools"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'scVI' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Probabilistic single-cell expression model","facets":{"areas":["single-cell"]},"id":"discovery-model-scvi","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-scib"}],"name":"scVI","source_ids":["src-discovery-scverse-scvi-tools"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"SegmentNT","version":null,"profile":{"summary":"SegmentNT: genomic sequence segmentation model","sections":[{"title":"Available evidence","body":"The discovery record links to instadeepai/nucleotide-transformer official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'SegmentNT' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Genomic sequence segmentation model","facets":{"areas":["genomics"]},"id":"discovery-model-segmentnt","kind":"model","links":[],"name":"SegmentNT","source_ids":["src-discovery-instadeepai-nucleotide-transformer"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":[],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"SpliceAI","version":null,"profile":{"summary":"SpliceAI predicts how sequence variants alter splice-site usage. Its variant annotation tool combines reference sequence with gene annotation.","sections":[{"title":"Architecture","body":"SpliceAI uses a residual convolutional network with dilated filters to integrate sequence context. It predicts donor, acceptor and non-splice-site probabilities along the sequence; comparing alleles converts those predictions into variant scores. The output is not tissue-specific.","source_ids":["src-discovery-illumina-spliceai","src-discovery-tkzeng-pangolin"],"source_locator":"SpliceAI README and linked Jaganathan et al. paper; Pangolin primary paper https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture, direct comparison with SpliceAI"},{"title":"How it works","body":"The tool evaluates reference and alternative alleles and reports predicted acceptor/donor gains and losses with their relative positions. Annotation, search distance and masking determine the reported variant scores.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"facts":[{"label":"Inputs","value":"VCF, reference FASTA and gene annotation","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"},{"label":"Outputs","value":"Acceptor/donor gain and loss delta scores","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"strengths":[{"text":"Produces splice-specific scores and predicted event positions without fitting a classifier to the user’s assay labels.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"limitations":[{"text":"Code, trained models and precomputed annotations have distinct use terms. Annotation and sequence checks can leave variants unscored.","source_ids":["src-discovery-illumina-spliceai"],"source_locator":"README.md: package description, License, Usage and Frequently Asked Questions"}],"diagram":{"title":"Conceptual procedure","steps":["Reference / alternate DNA","Dilated residual convolutions","Acceptor / donor probabilities","Allelic difference","Variant delta scores"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-illumina-spliceai","src-discovery-tkzeng-pangolin"],"source_locator":"SpliceAI README and linked Jaganathan et al. paper; Pangolin primary paper https://doi.org/10.1186/s13059-022-02664-4, Methods: Deep neural network architecture, direct comparison with SpliceAI"},"coverage":"reviewed","gaps":["An immutable hash for the exact 1.3.1 weight files has not been attached to this family record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Splicing effect prediction","facets":{"areas":["genomics"]},"id":"discovery-model-spliceai","kind":"model","links":[],"name":"SpliceAI","source_ids":["src-discovery-illumina-spliceai"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-virtual-cell-challenge-2026"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"STATE","version":null,"profile":{"summary":"STATE: cell state and perturbation modelling","sections":[{"title":"Available evidence","body":"The discovery record links to ArcInstitute/state official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-arcinstitute-state"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-arcinstitute-state"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'STATE' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Cell state and perturbation modelling","facets":{"areas":["single-cell"]},"id":"discovery-model-state","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-virtual-cell-challenge-2026"}],"name":"STATE","source_ids":["src-discovery-arcinstitute-state"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-glycanml"],"entity_level":"family","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"SweetNet","version":null,"profile":{"summary":"SweetNet: glycan graph learning model","sections":[{"title":"Available evidence","body":"The discovery record links to BojarLab/glycowork official source. Its scope is the recorded project or method; this entry does not establish that a named benchmark has been run.","source_ids":["src-discovery-bojarlab-glycowork"],"source_locator":"README.md"}],"facts":[],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-bojarlab-glycowork"],"source_locator":"README.md"}],"coverage":"limited","gaps":["The linked README.md has not yielded a reviewed architecture or procedure for the configuration 'SweetNet' in this release.","Unresolved registry fields: checkpoint (unextracted), code licence (unextracted), parameters (unextracted), training cutoff (unextracted), training data (unextracted), version (unextracted), weights licence (unextracted).","This family entry does not identify an evaluated checkpoint. Results from similarly named records are not automatically assigned here."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Glycan graph learning model","facets":{"areas":["glycomics"]},"id":"discovery-model-sweetnet","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-glycanml"}],"name":"SweetNet","source_ids":["src-discovery-bojarlab-glycowork"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"Bepler","version":null,"profile":{"summary":"TAPE Bepler is the method recorded for TAPE Fluorescence Bepler leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho; README.md > Leaderboard > Stability; row Bepler; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho; README.md > Leaderboard > Stability; row Bepler; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho; README.md > Leaderboard > Stability; row Bepler; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho; README.md > Leaderboard > Stability; row Bepler; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho; README.md > Leaderboard > Stability; row Bepler; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-bepler","kind":"model","links":[],"name":"TAPE Bepler","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"LSTM","version":null,"profile":{"summary":"TAPE LSTM is the method recorded for TAPE Fluorescence LSTM leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho; README.md > Leaderboard > Stability; row LSTM; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho; README.md > Leaderboard > Stability; row LSTM; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho; README.md > Leaderboard > Stability; row LSTM; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho; README.md > Leaderboard > Stability; row LSTM; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho; README.md > Leaderboard > Stability; row LSTM; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-lstm","kind":"model","links":[],"name":"TAPE LSTM","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"One Hot","version":null,"profile":{"summary":"TAPE One Hot is the method recorded for TAPE Fluorescence One Hot leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho; README.md > Leaderboard > Stability; row One Hot; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho; README.md > Leaderboard > Stability; row One Hot; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho; README.md > Leaderboard > Stability; row One Hot; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho; README.md > Leaderboard > Stability; row One Hot; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho; README.md > Leaderboard > Stability; row One Hot; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-one-hot","kind":"model","links":[],"name":"TAPE One Hot","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"ResNet","version":null,"profile":{"summary":"TAPE ResNet is the method recorded for TAPE Fluorescence ResNet leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho; README.md > Leaderboard > Stability; row ResNet; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho; README.md > Leaderboard > Stability; row ResNet; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho; README.md > Leaderboard > Stability; row ResNet; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho; README.md > Leaderboard > Stability; row ResNet; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho; README.md > Leaderboard > Stability; row ResNet; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-resnet","kind":"model","links":[],"name":"TAPE ResNet","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"Transformer","version":null,"profile":{"summary":"TAPE Transformer is the method recorded for TAPE Fluorescence Transformer leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho; README.md > Leaderboard > Stability; row Transformer; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho; README.md > Leaderboard > Stability; row Transformer; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho; README.md > Leaderboard > Stability; row Transformer; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho; README.md > Leaderboard > Stability; row Transformer; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho; README.md > Leaderboard > Stability; row Transformer; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-transformer","kind":"model","links":[],"name":"TAPE Transformer","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"entity_level":"method","missing_metadata":{"checkpoint":"unreported","version":"unreported"},"reported_name":"Unirep","version":null,"profile":{"summary":"TAPE Unirep is the method recorded for TAPE Fluorescence Unirep leaderboard evaluation. This page preserves the configuration reported by songlab-cal/tape official source.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TAPE Fluorescence TAPE Stability","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho; README.md > Leaderboard > Stability; row Unirep; column Spearman's rho"}],"facts":[{"label":"Recorded dataset","value":"TAPE Fluorescence source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho; README.md > Leaderboard > Stability; row Unirep; column Spearman's rho"},{"label":"Recorded dataset","value":"TAPE Stability source dataset","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho; README.md > Leaderboard > Stability; row Unirep; column Spearman's rho"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["src-discovery-songlab-cal-tape"],"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho; README.md > Leaderboard > Stability; row Unirep; column Spearman's rho"}],"coverage":"limited","gaps":["The retained evidence location (README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho; README.md > Leaderboard > Stability; row Unirep; column Spearman's rho) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint (unreported), version (unreported)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}},"description":"Method identity as printed in the historical TAPE leaderboard.","facets":{"areas":["protein-function"]},"id":"discovery-model-tape-unirep","kind":"model","links":[],"name":"TAPE Unirep","source_ids":["src-discovery-songlab-cal-tape"],"status":"discovered"} {"attributes":{"access":"official_source_linked","benchmark_applicability":"candidate; not evidence of a reported evaluation","candidate_benchmark_ids":["discovery-benchmark-beacon"],"entity_level":"method","missing_metadata":{"checkpoint":"unextracted","code_licence":"unextracted","parameters":"unextracted","training_cutoff":"unextracted","training_data":"unextracted","version":"unextracted","weights_licence":"unextracted"},"reported_name":"ViennaRNA RNAfold","version":null,"profile":{"summary":"RNAfold predicts RNA secondary structure using thermodynamic calculations in the ViennaRNA package.","sections":[{"title":"How it works","body":"An RNA sequence is processed using the selected energy model. The program calculates a minimum-free-energy secondary structure and can compute a partition function.","source_ids":["src-discovery-viennarna-viennarna"],"source_locator":"README.md: Program descriptions, RNAfold row"}],"facts":[{"label":"Method class","value":"Thermodynamic RNA folding","source_ids":["src-discovery-viennarna-viennarna"],"source_locator":"README.md: Program descriptions, RNAfold row"}],"strengths":[{"text":"Provides a mechanistic reference for comparison with learned RNA-structure predictors.","source_ids":["src-discovery-viennarna-viennarna"],"source_locator":"README.md: Program descriptions, RNAfold row"}],"limitations":[{"text":"The energy parameterisation and folding options belong to the evaluated configuration; secondary structure is not a complete 3D structure.","source_ids":["src-discovery-viennarna-viennarna"],"source_locator":"README.md: Program descriptions, RNAfold row"}],"diagram":{"title":"Conceptual procedure","steps":["RNA sequence","Thermodynamic model","Energy / partition calculation","Secondary-structure prediction"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["src-discovery-viennarna-viennarna"],"source_locator":"README.md: Program descriptions, RNAfold row"},"coverage":"reviewed","gaps":["An immutable model/checkpoint artifact and complete training-data manifest have not been extracted for this record."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}},"description":"Thermodynamic RNA secondary structure prediction","facets":{"areas":["rna"]},"id":"discovery-model-viennarna-rnafold","kind":"model","links":[{"relation":"applicable_to","target_id":"discovery-benchmark-beacon"}],"name":"ViennaRNA RNAfold","source_ids":["src-discovery-viennarna-viennarna"],"status":"discovered"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.33","printed_value":"0.33","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row Bepler; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-bepler-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-bepler-leaderboard-evaluation"}],"name":"TAPE Fluorescence Bepler Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.67","printed_value":"0.67","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row LSTM; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-lstm-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-lstm-leaderboard-evaluation"}],"name":"TAPE Fluorescence LSTM Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.14","printed_value":"0.14","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row One Hot; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-one-hot-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-one-hot-leaderboard-evaluation"}],"name":"TAPE Fluorescence One Hot Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.21","printed_value":"0.21","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row ResNet; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-resnet-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-resnet-leaderboard-evaluation"}],"name":"TAPE Fluorescence ResNet Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.68","printed_value":"0.68","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row Transformer; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-transformer-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-transformer-leaderboard-evaluation"}],"name":"TAPE Fluorescence Transformer Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.67","printed_value":"0.67","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Fluorescence; row Unirep; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-fluorescence-unirep-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-fluorescence-unirep-leaderboard-evaluation"}],"name":"TAPE Fluorescence Unirep Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.64","printed_value":"0.64","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row Bepler; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-bepler-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-bepler-leaderboard-evaluation"}],"name":"TAPE Stability Bepler Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.69","printed_value":"0.69","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row LSTM; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-lstm-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-lstm-leaderboard-evaluation"}],"name":"TAPE Stability LSTM Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.19","printed_value":"0.19","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row One Hot; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-one-hot-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-one-hot-leaderboard-evaluation"}],"name":"TAPE Stability One Hot Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.73","printed_value":"0.73","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row ResNet; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-resnet-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-resnet-leaderboard-evaluation"}],"name":"TAPE Stability ResNet Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.73","printed_value":"0.73","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row Transformer; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-transformer-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-transformer-leaderboard-evaluation"}],"name":"TAPE Stability Transformer Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"attributes":{"metric":"Spearman's rho","metric_direction":"higher","missing_metadata":{"denominator":"unextracted","seeds":"unreported","uncertainty":"unreported"},"numeric_value":"0.73","printed_value":"0.73","review":{"method":"automated table parse plus separate source read and manually enumerated row/value cross-check","notes":"All six rows checked for this task. Source checked, not experiment reproduced. Historical README does not resolve exact checkpoint or data release.","reviewed_at":"2026-09-16T10:35:25.806835+00:00","reviewer":"Codex research agent; no human review claimed"},"source_locator":"README.md > Leaderboard > Stability; row Unirep; column Spearman's rho","uncertainty":null,"unit":"correlation"},"description":"Reported score transcribed from the complete historical task table.","facets":{"areas":["protein-function"]},"id":"discovery-result-tape-stability-unirep-spearman-rho","kind":"result","links":[{"relation":"evaluation","target_id":"discovery-evaluation-tape-stability-unirep-leaderboard-evaluation"}],"name":"TAPE Stability Unirep Spearman rho","source_ids":["src-discovery-songlab-cal-tape"],"status":"source_checked"} {"id":"dna-foundation-models-2025","kind":"source","name":"Benchmarking DNA foundation models for genomic and genetic tasks","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","version":"PMC12663285.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1038/s41467-025-65823-8","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"5d8ca9bcf88cc1b38ad667906a2e4699b1aefa6d31c6f49259784930353f3202","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12663285/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:35.327Z","legacy_paper":{"id":"dna-foundation-models-2025","title":"Benchmarking DNA foundation models for genomic and genetic tasks","year":2025,"publication_status":"peer_reviewed","version":"PMC12663285.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Nature Communications; PMC ID: PMC12663285.","doi":"10.1038/s41467-025-65823-8"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"dnabert2-enhancer-2025","kind":"source","name":"Utilizing a deep learning model based on BERT for identifying enhancers and their strength","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11981215/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1371/journal.pone.0320085","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"d052b80efe7bfc1380994ad28503a5575f04ef940f74d5c9c137cb4ba6827863","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11981215/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558204+00:00","legacy_paper":{"id":"dnabert2-enhancer-2025","title":"Utilizing a deep learning model based on BERT for identifying enhancers and their strength","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11981215/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1371/journal.pone.0320085","notes":"Numeric result checked against Table 4 in primary full-text XML; journal/source: PLOS One."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"dnalongbench-2025","kind":"source","name":"DNALongBench: A Benchmark Suite for Long-Range DNA Prediction Tasks","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11741265/","version":"PMC11741265.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2025.01.06.631595","publication_status":"preprint","year":2025,"artifact_sha256":"fa440a17cecf16a5d872d50a30910f7591b5f6f78e10a944c6bda5ea8d7e32dd","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11741265/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.492545+00:00","legacy_paper":{"id":"dnalongbench-2025","title":"DNALongBench: A Benchmark Suite for Long-Range DNA Prediction Tasks","year":2025,"publication_status":"preprint","version":"PMC11741265.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11741265/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC11741265.","doi":"10.1101/2025.01.06.631595"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"eden-genomic-classification-2026","kind":"source","name":"EDEN: multiscale expected density of nucleotide encoding for enhanced DNA sequence classification with hybrid deep learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12879454/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1186/s12859-026-06367-6","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"38a6e26b3caffe8e021a2b0b672218e783aca9ee42046765e323946813015e65","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12879454/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:37.531Z","legacy_paper":{"id":"eden-genomic-classification-2026","title":"EDEN: multiscale expected density of nucleotide encoding for enhanced DNA sequence classification with hybrid deep learning","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12879454/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1186/s12859-026-06367-6","notes":"DNABERT-2 comparator 70.52 is printed in Table 5. The article does not clearly document whether this comparator was independently rerun or consolidated from prior GUE results, so evaluation origin is conservatively marked paper_compilation."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"enbed-2024","kind":"source","name":"Understanding the natural language of DNA using encoder–decoder foundation models with byte-level precision","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1093/bioadv/vbae117","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"e95d4be70d32e61af5a92eda8ea66f25a2cc629e3f83e7b5241b13cde8bdb83b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:35.378Z","legacy_paper":{"id":"enbed-2024","title":"Understanding the natural language of DNA using encoder–decoder foundation models with byte-level precision","year":2024,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Bioinformatics Advances; PMC ID: PMC11341122.","doi":"10.1093/bioadv/vbae117"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"enhancer-position-encoding-2024","kind":"source","name":"A deep learning model for DNA enhancer prediction based on nucleotide position aware feature encoding","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11167433/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1016/j.isci.2024.110030","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"0183b6a111b1b02344cad35a571a1fd2c56257e406c5be1df69f7902c5d06749","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11167433/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"enhancer-position-encoding-2024","title":"A deep learning model for DNA enhancer prediction based on nucleotide position aware feature encoding","year":2024,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11167433/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: iScience; PMC ID: PMC11167433. Task-specific CNN baseline, included as a DNA benchmark protocol reference.","doi":"10.1016/j.isci.2024.110030"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ensemble-idp-docking-2025","kind":"source","name":"Ensemble docking for intrinsically disordered proteins","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11785235/","version":"preprint archived 2025-01-26","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2025.01.23.634614","publication_status":"preprint","year":2025,"artifact_sha256":"d02d91cdde41cb76ec5c86b532dffc564879c69e764a8c6b7752460fbbfd24b7","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11785235/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:56.275Z","legacy_paper":{"id":"ensemble-idp-docking-2025","title":"Ensemble docking for intrinsically disordered proteins","year":2025,"publication_status":"preprint","version":"preprint archived 2025-01-26","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11785235/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11785235.","doi":"10.1101/2025.01.23.634614"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ernie-rna-2025","kind":"source","name":"ERNIE-RNA: an RNA language model with structure-enhanced representations","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12627772/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s41467-025-64972-0","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"0bd1d4b3cbf5d59d452cec4864614947861efcee050ba07e7de395cd90630047","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12627772/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558206+00:00","legacy_paper":{"id":"ernie-rna-2025","title":"ERNIE-RNA: an RNA language model with structure-enhanced representations","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12627772/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s41467-025-64972-0","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Nature Communications."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"esm2-amp-2025","kind":"source","name":"ESM2_AMP: an interpretable framework for protein–protein interactions prediction and biological mechanism discovery","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12392411/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/bib/bbaf434","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"8e7ad6efb72ca28d73037cdf465b0e62f99cd6d0ee4ca9eaf96a4c48da22fd6c","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12392411/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:58.625Z","legacy_paper":{"id":"esm2-amp-2025","title":"ESM2_AMP: an interpretable framework for protein–protein interactions prediction and biological mechanism discovery","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12392411/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Briefings in Bioinformatics; PMC ID: PMC12392411. Paper has multiple model variants; selected named ESM2_AMPS variant only.","doi":"10.1093/bib/bbaf434"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"esm2-ofs-fitness-2025","kind":"source","name":"Pseudo-perplexity in One Fell Swoop for Protein Fitness Estimation","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://journals.aps.org/prxlife/pdf/10.1103/zhx7-hcmm","version":"PRX Life 2025 journal article","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1103/zhx7-hcmm","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"085ef646f11b8e5335c4b3d86b15fb6c7bf5edf4a80a8b622753ac69d9991a67","artifact_url":"https://harvest.aps.org/v2/journals/articles/10.1103/zhx7-hcmm/fulltext","artifact_retrieved_at":"2026-09-16T10:45:41.099916+00:00","legacy_paper":{"id":"esm2-ofs-fitness-2025","title":"Pseudo-perplexity in One Fell Swoop for Protein Fitness Estimation","year":2025,"publication_status":"peer_reviewed","version":"PRX Life 2025 journal article","source_url":"https://journals.aps.org/prxlife/pdf/10.1103/zhx7-hcmm","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1103/zhx7-hcmm","notes":"Final journal Table I, ESM2: OFS PP Aggregate Mean 0.403 checked directly; manuscript PMC11257618 printed the same value. Other models in the table are imported ProteinGym baselines; this row is the authors’ own evaluation."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-2ome-lm-2025","kind":"evaluation","name":"2OMe-LM: human RNA 2-prime-O-methylation site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["human RNA 2-prime-O-methylation site prediction"]},"source_ids":["2ome-lm-2025"],"links":[{"relation":"model","target_id":"reported-model-7f5b8234967c54"},{"relation":"benchmark","target_id":"reported-task-82fc7843f07324"},{"relation":"dataset","target_id":"reported-dataset-bd3d8e7d6cd196"}],"attributes":{"origin":"author_reported","protocol":"pretrained RNA language model predictor","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"5-fold cross-validation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-antibody-deamidation-plm-2024","kind":"evaluation","name":"ESM-2 650M embeddings + classifier: antibody deamidation-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["antibody deamidation-site prediction"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[{"relation":"model","target_id":"reported-model-4921459942b45f"},{"relation":"benchmark","target_id":"reported-task-0647b0364def8f"},{"relation":"dataset","target_id":"reported-dataset-0edd8f724db696"}],"attributes":{"origin":"author_reported","protocol":"global contextual embeddings only","version":"esm2_t33_650m_UR50D","comparison":{"protocol_id":null,"dataset_version":null,"split":"fivefold stratified CV","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-barcodebert-2026","kind":"evaluation","name":"BarcodeBERT (4–4-4): unseen-species genus classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["unseen-species genus classification"]},"source_ids":["barcodebert-2026"],"links":[{"relation":"model","target_id":"reported-model-05103f72325fe5"},{"relation":"benchmark","target_id":"reported-task-4a54ce01b5a855"},{"relation":"dataset","target_id":"reported-dataset-bc127dc9c441fe"}],"attributes":{"origin":"author_reported","protocol":"genus-level nearest-neighbor probe on species unseen in training","version":"4–4–4","comparison":{"protocol_id":null,"dataset_version":null,"split":"1-NN probe","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-birna-bert-2025","kind":"evaluation","name":"BiRNA-BERT: extremely long RNA species classification","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["extremely long RNA species classification"]},"source_ids":["birna-bert-2025"],"links":[{"relation":"model","target_id":"reported-model-d3fd83835a2d44"},{"relation":"benchmark","target_id":"reported-task-c40dac20d9af66"},{"relation":"dataset","target_id":"reported-dataset-ebc3f5fda43972"}],"attributes":{"origin":"author_reported","protocol":"adaptive tokenization on full-length long RNA sequences","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-cathe2-2025","kind":"evaluation","name":"CATHe2 + ProstT5: CATH superfamily annotation","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["CATH superfamily annotation"]},"source_ids":["cathe2-2025"],"links":[{"relation":"model","target_id":"reported-model-49bc768f46b366"},{"relation":"benchmark","target_id":"reported-task-c98e91ffc7247d"},{"relation":"dataset","target_id":"reported-dataset-6e0c28dfde7337"}],"attributes":{"origin":"author_reported","protocol":"amino-acid and structural alphabet embedding classifier","version":"full ProstT5","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-clathrin-plm-2025","kind":"evaluation","name":"ESM-2 embedding + paper classifier: clathrin protein classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["clathrin protein classification"]},"source_ids":["clathrin-plm-2025"],"links":[{"relation":"model","target_id":"reported-model-e4710b1c3facf2"},{"relation":"benchmark","target_id":"reported-task-786c09824e9bf5"},{"relation":"dataset","target_id":"reported-dataset-0aab382ca2c063"}],"attributes":{"origin":"independent_paper","protocol":"single-feature ESM-2 embedding comparison","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-cobra-rna-binding-2026","kind":"evaluation","name":"ERNIE-RNA + CoBRA: RNA compound-binding site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA compound-binding site prediction"]},"source_ids":["cobra-rna-binding-2026"],"links":[{"relation":"model","target_id":"reported-model-7ad28cd57f5f5b"},{"relation":"benchmark","target_id":"reported-task-3a3bff34cce634"},{"relation":"dataset","target_id":"reported-dataset-b1af840b76b351"}],"attributes":{"origin":"author_reported","protocol":"ERNIE-RNA embedding with TCL focal loss","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-codonbert-vaccines-2024","kind":"evaluation","name":"CodonBERT: flu-vaccine mRNA property prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["flu-vaccine mRNA property prediction"]},"source_ids":["codonbert-vaccines-2024"],"links":[{"relation":"model","target_id":"reported-model-cd246741c378db"},{"relation":"benchmark","target_id":"reported-task-1c74661df2c401"},{"relation":"dataset","target_id":"reported-dataset-54b9bc432928d6"}],"attributes":{"origin":"author_reported","protocol":"codon-based model fine-tuned for downstream regression","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-dart-eval-regulatory-2024","kind":"evaluation","name":"DNABERT-2: regulatory element identification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory element identification"]},"source_ids":["dart-eval-regulatory-2024"],"links":[{"relation":"model","target_id":"reported-model-28413ae1766316"},{"relation":"benchmark","target_id":"reported-task-cdbee1c9285568"},{"relation":"dataset","target_id":"reported-dataset-b6ce37ba678d39"}],"attributes":{"origin":"independent_paper","protocol":"zero-shot likelihood ranking: higher likelihood for cCRE than matched control","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-dnabert2-enhancer-2025","kind":"evaluation","name":"DNABERT2-Enhancer: enhancer recognition","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer recognition"]},"source_ids":["dnabert2-enhancer-2025"],"links":[{"relation":"model","target_id":"reported-model-d0e594ec3c0430"},{"relation":"benchmark","target_id":"reported-task-86a628af87ff8f"},{"relation":"dataset","target_id":"reported-dataset-6212e779949708"}],"attributes":{"origin":"author_reported","protocol":"first-layer enhancer versus non-enhancer classifier","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"5-fold cross-validation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-eden-genomic-classification-2026","kind":"evaluation","name":"DNABERT-2: human core-promoter classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["human core-promoter classification"]},"source_ids":["eden-genomic-classification-2026"],"links":[{"relation":"model","target_id":"reported-model-2cb8118b4c77c0"},{"relation":"benchmark","target_id":"reported-task-9f62e739c6371e"},{"relation":"dataset","target_id":"reported-dataset-8e9488896becd4"}],"attributes":{"origin":"paper_compilation","protocol":"DNABERT-2 comparator in consolidated H-CPD table; rerun provenance not explicit","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-ernie-rna-2025","kind":"evaluation","name":"ERNIE-RNA: RNA secondary-structure prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary-structure prediction"]},"source_ids":["ernie-rna-2025"],"links":[{"relation":"model","target_id":"reported-model-d023fbe78bc4df"},{"relation":"benchmark","target_id":"reported-task-a2bf7ddbc71d23"},{"relation":"dataset","target_id":"reported-dataset-abdfba8cce7486"}],"attributes":{"origin":"author_reported","protocol":"zero-shot attention-derived base-pair prediction","version":"86M","comparison":{"protocol_id":null,"dataset_version":null,"split":"cross-family test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-esm2-ofs-fitness-2025","kind":"evaluation","name":"ESM2 OFS pseudo-perplexity: protein variant fitness prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein variant fitness prediction"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[{"relation":"model","target_id":"reported-model-40ce004270dee4"},{"relation":"benchmark","target_id":"reported-task-c7a8a372f77886"},{"relation":"dataset","target_id":"reported-dataset-9c186c8f4ed3f4"}],"attributes":{"origin":"author_reported","protocol":"authors’ zero-shot ESM2 OFS pseudo-perplexity evaluation; aggregate mean across ProteinGym substitution assays","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"aggregate across assays","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-fusion-breakpoint-foundation-models-2026","kind":"evaluation","name":"Nucleotide Transformer + NN (middle): gene fusion breakpoint classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["gene fusion breakpoint classification"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[{"relation":"model","target_id":"reported-model-1a67087ac262c5"},{"relation":"benchmark","target_id":"reported-task-ee34721cf55590"},{"relation":"dataset","target_id":"reported-dataset-f6922a9744ba27"}],"attributes":{"origin":"independent_paper","protocol":"middle embedding with neural-network classifier","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"full test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-genomic-tokenizer-selection-2025","kind":"evaluation","name":"Caduceus (character tokens): regulatory sequence classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory sequence classification"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[{"relation":"model","target_id":"reported-model-d5bc536ca6f0d3"},{"relation":"benchmark","target_id":"reported-task-cd127e56fb1f04"},{"relation":"dataset","target_id":"reported-dataset-0bba1c9a7ae410"}],"attributes":{"origin":"independent_paper","protocol":"task-category MCC across benchmark datasets","version":"3.9M parameter variant","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper benchmark summary","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-gsmformer-ppi-2026","kind":"evaluation","name":"GSMFormer-PPI + ProstT5: protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein interaction prediction"]},"source_ids":["gsmformer-ppi-2026"],"links":[{"relation":"model","target_id":"reported-model-73ae07fb5be204"},{"relation":"benchmark","target_id":"reported-task-dfa8f2285dbfa5"},{"relation":"dataset","target_id":"reported-dataset-07d355c146be1f"}],"attributes":{"origin":"author_reported","protocol":"ProstT5 embeddings as graph node features","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-megsite-2025","kind":"evaluation","name":"MegSite + ESM3: DNA-binding residue prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["DNA-binding residue prediction"]},"source_ids":["megsite-2025"],"links":[{"relation":"model","target_id":"reported-model-86393c76dd8fa9"},{"relation":"benchmark","target_id":"reported-task-9917a0e69f33e7"},{"relation":"dataset","target_id":"reported-dataset-739aee3cf8d6f1"}],"attributes":{"origin":"author_reported","protocol":"ESM3 multimodal embedding ablation in MegSite","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-mrna-lm-2025","kind":"evaluation","name":"mRNA-LM: mRNA half-life prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA half-life prediction"]},"source_ids":["mrna-lm-2025"],"links":[{"relation":"model","target_id":"reported-model-54d974e8e08043"},{"relation":"benchmark","target_id":"reported-task-5693847493f19f"},{"relation":"dataset","target_id":"reported-dataset-52f00ccaabf0d9"}],"attributes":{"origin":"author_reported","protocol":"average test performance across cross-validation splits","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set across CV splits","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-mrnabert-2025","kind":"evaluation","name":"mRNABERT: translation-efficiency prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["translation-efficiency prediction"]},"source_ids":["mrnabert-2025"],"links":[{"relation":"model","target_id":"reported-model-13bd2a6c2d8178"},{"relation":"benchmark","target_id":"reported-task-f7142c3b3e0f3c"},{"relation":"dataset","target_id":"reported-dataset-1744719eef145b"}],"attributes":{"origin":"author_reported","protocol":"human translation-efficiency regression at 3066-nt input","version":"3066-nt input","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-mulan-2025","kind":"evaluation","name":"MULAN-ESM2 S: human protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["human protein-protein interaction prediction"]},"source_ids":["mulan-2025"],"links":[{"relation":"model","target_id":"reported-model-a29203c09857ef"},{"relation":"benchmark","target_id":"reported-task-6e54c7452b2b81"},{"relation":"dataset","target_id":"reported-dataset-38151fa548e291"}],"attributes":{"origin":"author_reported","protocol":"MULAN sequence-structure model based on ESM2 8M","version":"small ESM2 backbone","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-phylogpn-2025","kind":"evaluation","name":"PhyloGPN: ClinVar 3-prime UTR variant classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["ClinVar 3-prime UTR variant classification"]},"source_ids":["phylogpn-2025"],"links":[{"relation":"model","target_id":"reported-model-cbbe04b826ceff"},{"relation":"benchmark","target_id":"reported-task-ed3dd3b83c4505"},{"relation":"dataset","target_id":"reported-dataset-a28180d33f7a23"}],"attributes":{"origin":"author_reported","protocol":"log-likelihood-ratio scoring","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-polya-glm-2025","kind":"evaluation","name":"HyenaDNA: polyadenylation site detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["polyadenylation site detection"]},"source_ids":["polya-glm-2025"],"links":[{"relation":"model","target_id":"reported-model-953007693fb72a"},{"relation":"benchmark","target_id":"reported-task-13dfe6b33e71ed"},{"relation":"dataset","target_id":"reported-dataset-55f200c9481409"}],"attributes":{"origin":"independent_paper","protocol":"few-shot Gene-Gene negative-set comparison","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"5-fold cross-validation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-rlsite-rna-binding-2025","kind":"evaluation","name":"RLsite: RNA-small-molecule binding-site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA-small-molecule binding-site prediction"]},"source_ids":["rlsite-rna-binding-2025"],"links":[{"relation":"model","target_id":"reported-model-52b9c685d99290"},{"relation":"benchmark","target_id":"reported-task-b00a636d1ed8d9"},{"relation":"dataset","target_id":"reported-dataset-1c7f8ebb1968d9"}],"attributes":{"origin":"author_reported","protocol":"RNA language-model plus graph-attention classifier","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-rnaret-2026","kind":"evaluation","name":"RNAret: miRNA-mRNA interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["miRNA-mRNA interaction prediction"]},"source_ids":["rnaret-2026"],"links":[{"relation":"model","target_id":"reported-model-7234658bc9c828"},{"relation":"benchmark","target_id":"reported-task-46e927bea10702"},{"relation":"dataset","target_id":"reported-dataset-99afd0c86b2954"}],"attributes":{"origin":"author_reported","protocol":"5-mer RNAret classifier; 72/8/20 train/validation/test split","version":"5-mer","comparison":{"protocol_id":null,"dataset_version":null,"split":"held-out test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-spin-protein-function-2026","kind":"evaluation","name":"SPIN + ESM2-35M: protein function annotation","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein function annotation"]},"source_ids":["spin-protein-function-2026"],"links":[{"relation":"model","target_id":"reported-model-f8f0257b98749a"},{"relation":"benchmark","target_id":"reported-task-c4a578065f44b2"},{"relation":"dataset","target_id":"reported-dataset-dba1707164d296"}],"attributes":{"origin":"author_reported","protocol":"frozen ESM2-35M backbone in SPIN","version":"ESM2-35M frozen","comparison":{"protocol_id":null,"dataset_version":null,"split":"test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-b2-structure-informed-plm-2025","kind":"evaluation","name":"structure-informed pLM: protein variant-effect classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein variant-effect classification"]},"source_ids":["structure-informed-plm-2025"],"links":[{"relation":"model","target_id":"reported-model-035a3ab36a3a6a"},{"relation":"benchmark","target_id":"reported-task-83be0998084c91"},{"relation":"dataset","target_id":"reported-dataset-2eaa2a051d45ee"}],"attributes":{"origin":"author_reported","protocol":"combined amino-acid, secondary structure, solvent accessibility and contact-map scoring","version":"not stated in table","comparison":{"protocol_id":null,"dataset_version":null,"split":"paper evaluation","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-001","kind":"evaluation","name":"Caduceus-Ph: Human 5mC detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"model","target_id":"reported-model-47521865af7b04"},{"relation":"benchmark","target_id":"reported-task-988ff78f86471e"},{"relation":"dataset","target_id":"reported-dataset-463197d6a98b99"}],"attributes":{"origin":"independent_paper","protocol":"Binary epigenetic-modification classification as reported in the paper.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-002","kind":"evaluation","name":"NT-v2: Human 5mC detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"model","target_id":"reported-model-3af86cb274f658"},{"relation":"benchmark","target_id":"reported-task-988ff78f86471e"},{"relation":"dataset","target_id":"reported-dataset-463197d6a98b99"}],"attributes":{"origin":"independent_paper","protocol":"Binary epigenetic-modification classification as reported in the paper.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-003","kind":"evaluation","name":"ENBED: Enhancer classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"model","target_id":"reported-model-8db190bee6aae5"},{"relation":"benchmark","target_id":"reported-task-132da895d4c381"},{"relation":"dataset","target_id":"reported-dataset-f0bf60a62ad7c3"}],"attributes":{"origin":"author_reported","protocol":"Reported Genomic Benchmarks classification accuracy.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-004","kind":"evaluation","name":"ENBED (GRCh38): Enhancer classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"model","target_id":"reported-model-bdb1db16d3389d"},{"relation":"benchmark","target_id":"reported-task-132da895d4c381"},{"relation":"dataset","target_id":"reported-dataset-f0bf60a62ad7c3"}],"attributes":{"origin":"author_reported","protocol":"ENBED trained on GRCh38; reported Genomic Benchmarks classification accuracy.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-005","kind":"evaluation","name":"DNABERT-2: G-quadruplex classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"model","target_id":"reported-model-ade36035f58f27"},{"relation":"benchmark","target_id":"reported-task-c9d2a6435979e9"},{"relation":"dataset","target_id":"reported-dataset-9e9d18bc5bfb8b"}],"attributes":{"origin":"independent_paper","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","version":"117M","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-006","kind":"evaluation","name":"Caduceus: G-quadruplex classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"model","target_id":"reported-model-abc19288fe9009"},{"relation":"benchmark","target_id":"reported-task-c9d2a6435979e9"},{"relation":"dataset","target_id":"reported-dataset-9e9d18bc5bfb8b"}],"attributes":{"origin":"independent_paper","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","version":"8M","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-007","kind":"evaluation","name":"HyenaDNA: Enhancer-target gene prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"model","target_id":"reported-model-9b3bc255532dd3"},{"relation":"benchmark","target_id":"reported-task-2cbac97dd849f5"},{"relation":"dataset","target_id":"reported-dataset-fcb5752d916d5d"}],"attributes":{"origin":"independent_paper","protocol":"Long-range ETGP benchmark; source table reports AUROC.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-008","kind":"evaluation","name":"Caduceus-Ph: Enhancer-target gene prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"model","target_id":"reported-model-a7cfacf25d97ad"},{"relation":"benchmark","target_id":"reported-task-2cbac97dd849f5"},{"relation":"dataset","target_id":"reported-dataset-fcb5752d916d5d"}],"attributes":{"origin":"independent_paper","protocol":"Long-range ETGP benchmark; source table reports AUROC.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-009","kind":"evaluation","name":"RiNALMo: Mean ribosome load from MPRA","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"model","target_id":"reported-model-3e58d0faf88d2e"},{"relation":"benchmark","target_id":"reported-task-57dc3dcdb67a81"},{"relation":"dataset","target_id":"reported-dataset-3a3e3880a3fed0"}],"attributes":{"origin":"independent_paper","protocol":"Linear probe; mean across ten random seeds.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-010","kind":"evaluation","name":"RNA-FM: Mean ribosome load from MPRA","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"model","target_id":"reported-model-43cf51abca83d1"},{"relation":"benchmark","target_id":"reported-task-57dc3dcdb67a81"},{"relation":"dataset","target_id":"reported-dataset-3a3e3880a3fed0"}],"attributes":{"origin":"independent_paper","protocol":"Linear probe; mean across ten random seeds.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-011","kind":"evaluation","name":"BPfold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"model","target_id":"reported-model-c464bface507ee"},{"relation":"benchmark","target_id":"reported-task-dc82fcbfb44935"},{"relation":"dataset","target_id":"reported-dataset-8317793f18b026"}],"attributes":{"origin":"author_reported","protocol":"Family-wise evaluation of canonical base-pair predictions.","version":null,"comparison":{"protocol_id":null,"dataset_version":"116 RNAs","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-012","kind":"evaluation","name":"RNAfold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"model","target_id":"reported-model-52eee4cc67ca26"},{"relation":"benchmark","target_id":"reported-task-dc82fcbfb44935"},{"relation":"dataset","target_id":"reported-dataset-8317793f18b026"}],"attributes":{"origin":"independent_paper","protocol":"Family-wise evaluation of canonical base-pair predictions.","version":null,"comparison":{"protocol_id":null,"dataset_version":"116 RNAs","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-013","kind":"evaluation","name":"TU-Fold (aug): RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"model","target_id":"reported-model-d1cd9a425f9bbd"},{"relation":"benchmark","target_id":"reported-task-5ec7581b246ea6"},{"relation":"dataset","target_id":"reported-dataset-f2e729f333a333"}],"attributes":{"origin":"author_reported","protocol":"Three-fold training and evaluation; source reports mean and standard deviation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-014","kind":"evaluation","name":"UFold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"model","target_id":"reported-model-e3abb0b9a2ec79"},{"relation":"benchmark","target_id":"reported-task-5ec7581b246ea6"},{"relation":"dataset","target_id":"reported-dataset-f2e729f333a333"}],"attributes":{"origin":"independent_paper","protocol":"Three-fold training and evaluation; source reports mean and standard deviation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-015","kind":"evaluation","name":"DEBFold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"model","target_id":"reported-model-3f850c08d76410"},{"relation":"benchmark","target_id":"reported-task-016f70615f2cfc"},{"relation":"dataset","target_id":"reported-dataset-18ebde58579c2b"}],"attributes":{"origin":"author_reported","protocol":"Median F1 on the prepared TestSetβ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-016","kind":"evaluation","name":"RNAfold: RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"model","target_id":"reported-model-ed7f0db85facb1"},{"relation":"benchmark","target_id":"reported-task-016f70615f2cfc"},{"relation":"dataset","target_id":"reported-dataset-18ebde58579c2b"}],"attributes":{"origin":"independent_paper","protocol":"Median F1 on the prepared TestSetβ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-017","kind":"evaluation","name":"ESM-2: Zero-shot substitution mutation effects: stability","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"model","target_id":"reported-model-d326e3c4e3ba20"},{"relation":"benchmark","target_id":"reported-task-6243658a1bc215"},{"relation":"dataset","target_id":"reported-dataset-7cec655cd742f3"}],"attributes":{"origin":"independent_paper","protocol":"Zero-shot mutation scores; average Spearman across stability-category assays.","version":"15B","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-018","kind":"evaluation","name":"ProteinMPNN: Zero-shot substitution mutation effects: stability","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"model","target_id":"reported-model-e0443048c6e110"},{"relation":"benchmark","target_id":"reported-task-6243658a1bc215"},{"relation":"dataset","target_id":"reported-dataset-7cec655cd742f3"}],"attributes":{"origin":"independent_paper","protocol":"Zero-shot mutation scores; average Spearman across stability-category assays.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-019","kind":"evaluation","name":"FUJISAN: Enzyme functional identity prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"model","target_id":"reported-model-9c10fbec02a365"},{"relation":"benchmark","target_id":"reported-task-1ebf9b408517f9"},{"relation":"dataset","target_id":"reported-dataset-5197cca532f89d"}],"attributes":{"origin":"author_reported","protocol":"Sequence and structural feature integration; paper-reported test sub-dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-020","kind":"evaluation","name":"ESM2: Enzyme functional identity prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"model","target_id":"reported-model-ccd1160ad4ec27"},{"relation":"benchmark","target_id":"reported-task-1ebf9b408517f9"},{"relation":"dataset","target_id":"reported-dataset-5197cca532f89d"}],"attributes":{"origin":"independent_paper","protocol":"Comparator evaluated on the paper-reported test sub-dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-021","kind":"evaluation","name":"ESM-2: Mutated RBD binding prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"model","target_id":"reported-model-d0d5df2beb02b2"},{"relation":"benchmark","target_id":"reported-task-00e594df6a182d"},{"relation":"dataset","target_id":"reported-dataset-becc215358afd0"}],"attributes":{"origin":"independent_paper","protocol":"Frozen mean-pooled representation with downstream regression; position-stratified split.","version":"8M","comparison":{"protocol_id":null,"dataset_version":null,"split":"position-stratified","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-022","kind":"evaluation","name":"ESM-C: Mutated RBD binding prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"model","target_id":"reported-model-f83c0b833411a7"},{"relation":"benchmark","target_id":"reported-task-00e594df6a182d"},{"relation":"dataset","target_id":"reported-dataset-becc215358afd0"}],"attributes":{"origin":"independent_paper","protocol":"Frozen mean-pooled representation with downstream regression; position-stratified split.","version":"300M","comparison":{"protocol_id":null,"dataset_version":null,"split":"position-stratified","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-023","kind":"evaluation","name":"PST: Zero-shot variant effect prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"model","target_id":"reported-model-7dd5992188a868"},{"relation":"benchmark","target_id":"reported-task-a5141363b0ee45"},{"relation":"dataset","target_id":"reported-dataset-bd9255afb783d6"}],"attributes":{"origin":"author_reported","protocol":"Zero-shot VEP; paper averages absolute Spearman correlations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-024","kind":"evaluation","name":"ESM-2: Zero-shot variant effect prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"model","target_id":"reported-model-d25dab1a9c4fff"},{"relation":"benchmark","target_id":"reported-task-a5141363b0ee45"},{"relation":"dataset","target_id":"reported-dataset-bd9255afb783d6"}],"attributes":{"origin":"independent_paper","protocol":"Zero-shot VEP; paper averages absolute Spearman correlations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-025","kind":"evaluation","name":"scGPT: Cell-type identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"model","target_id":"reported-model-3bdb3093e8d531"},{"relation":"benchmark","target_id":"reported-task-5b929593eefc76"},{"relation":"dataset","target_id":"reported-dataset-488d5de6bb9c1b"}],"attributes":{"origin":"independent_paper","protocol":"Native scLLM cell-type identification as reported in Table 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-026","kind":"evaluation","name":"Geneformer: Cell-type identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"model","target_id":"reported-model-b46ae14b9927ac"},{"relation":"benchmark","target_id":"reported-task-5b929593eefc76"},{"relation":"dataset","target_id":"reported-dataset-488d5de6bb9c1b"}],"attributes":{"origin":"independent_paper","protocol":"Native scLLM cell-type identification as reported in Table 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-027","kind":"evaluation","name":"C2S (GPT-2 Large): Combinatorial cell-label classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"model","target_id":"reported-model-ab02228f50a37c"},{"relation":"benchmark","target_id":"reported-task-7efe245cc94ee5"},{"relation":"dataset","target_id":"reported-dataset-87e91d9d6e6f4b"}],"attributes":{"origin":"author_reported","protocol":"Partial-credit labels including cell type, perturbation, and dose.","version":"GPT-2 Large","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-028","kind":"evaluation","name":"Geneformer: Combinatorial cell-label classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"model","target_id":"reported-model-06816ce9073144"},{"relation":"benchmark","target_id":"reported-task-7efe245cc94ee5"},{"relation":"dataset","target_id":"reported-dataset-87e91d9d6e6f4b"}],"attributes":{"origin":"independent_paper","protocol":"Partial-credit labels including cell type, perturbation, and dose.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-029","kind":"evaluation","name":"scGPT: Cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"model","target_id":"reported-model-77ad27d4098177"},{"relation":"benchmark","target_id":"reported-task-660753ec94e631"},{"relation":"dataset","target_id":"reported-dataset-8f123f006964ad"}],"attributes":{"origin":"paper_compilation","protocol":"Zero-shot setting; source caption says some comparator rows come from GenePT.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-030","kind":"evaluation","name":"Geneformer: Cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"model","target_id":"reported-model-d60f505aabb19c"},{"relation":"benchmark","target_id":"reported-task-660753ec94e631"},{"relation":"dataset","target_id":"reported-dataset-8f123f006964ad"}],"attributes":{"origin":"paper_compilation","protocol":"Zero-shot setting; source caption says some comparator rows come from GenePT.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-031","kind":"evaluation","name":"scRegNet (Geneformer backbone): Gene-regulatory link prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"model","target_id":"reported-model-60455ff7cc0c15"},{"relation":"benchmark","target_id":"reported-task-3063ed4da76b4b"},{"relation":"dataset","target_id":"reported-dataset-2ad2fad5e1cd0a"}],"attributes":{"origin":"author_reported","protocol":"TFs plus 500 variable genes; mean from 50 independent evaluations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-032","kind":"evaluation","name":"scRegNet (scBERT backbone): Gene-regulatory link prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"model","target_id":"reported-model-89f5a8f309fa18"},{"relation":"benchmark","target_id":"reported-task-3063ed4da76b4b"},{"relation":"dataset","target_id":"reported-dataset-2ad2fad5e1cd0a"}],"attributes":{"origin":"author_reported","protocol":"TFs plus 500 variable genes; mean from 50 independent evaluations.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-033","kind":"evaluation","name":"ProkBERT-mini: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"model","target_id":"reported-model-4438513d9cd42c"},{"relation":"benchmark","target_id":"reported-task-3891811dcce8b3"},{"relation":"dataset","target_id":"reported-dataset-48def1da574597"}],"attributes":{"origin":"author_reported","protocol":"Promoter versus non-promoter classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-034","kind":"evaluation","name":"Promotech: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"model","target_id":"reported-model-0d147487bf97be"},{"relation":"benchmark","target_id":"reported-task-3891811dcce8b3"},{"relation":"dataset","target_id":"reported-dataset-48def1da574597"}],"attributes":{"origin":"independent_paper","protocol":"Promoter versus non-promoter classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-035","kind":"evaluation","name":"Eco70PromBERT: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"model","target_id":"reported-model-5b70fccb70bb70"},{"relation":"benchmark","target_id":"reported-task-e5c34f686ac403"},{"relation":"dataset","target_id":"reported-dataset-a1da4a37eb46a5"}],"attributes":{"origin":"author_reported","protocol":"BERT-base with 1bp tokenizer; 110 promoters and 108 non-promoters.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-036","kind":"evaluation","name":"iPro70-FMWin: E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"model","target_id":"reported-model-23cb15b93c00ff"},{"relation":"benchmark","target_id":"reported-task-e5c34f686ac403"},{"relation":"dataset","target_id":"reported-dataset-a1da4a37eb46a5"}],"attributes":{"origin":"independent_paper","protocol":"Compared on the same independent test dataset; 110 promoters and 108 non-promoters.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-037","kind":"evaluation","name":"EVO2: Genome-wide prophage detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"model","target_id":"reported-model-aa763db2cfdeff"},{"relation":"benchmark","target_id":"reported-task-dd001540e0f4ec"},{"relation":"dataset","target_id":"reported-dataset-1b4f6ea24c0587"}],"attributes":{"origin":"independent_paper","protocol":"Genomic language model fine-tuned for prophage detection; genome-wide evaluation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-038","kind":"evaluation","name":"geNomad: Genome-wide prophage detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"model","target_id":"reported-model-d0677d52d2b9fd"},{"relation":"benchmark","target_id":"reported-task-dd001540e0f4ec"},{"relation":"dataset","target_id":"reported-dataset-1b4f6ea24c0587"}],"attributes":{"origin":"independent_paper","protocol":"Traditional specialist comparator; genome-wide evaluation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-039","kind":"evaluation","name":"NABAS+: Metagenomic taxonomic classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"model","target_id":"reported-model-e7d203bd99ca99"},{"relation":"benchmark","target_id":"reported-task-92137759a9e7b0"},{"relation":"dataset","target_id":"reported-dataset-b462aa24561fba"}],"attributes":{"origin":"author_reported","protocol":"Newly generated sample19 used for classifier comparison.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-040","kind":"evaluation","name":"MetaPhlAn3: Metagenomic taxonomic classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"model","target_id":"reported-model-df4084611520b7"},{"relation":"benchmark","target_id":"reported-task-92137759a9e7b0"},{"relation":"dataset","target_id":"reported-dataset-b462aa24561fba"}],"attributes":{"origin":"independent_paper","protocol":"Newly generated sample19 used for classifier comparison.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-041","kind":"evaluation","name":"Chai-1: Lipid–protein binding pose","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"model","target_id":"reported-model-eae60780097101"},{"relation":"benchmark","target_id":"reported-task-ff2dec63c5a3dd"},{"relation":"dataset","target_id":"reported-dataset-6a44f5946cd7ab"}],"attributes":{"origin":"independent_paper","protocol":"Top-scoring pose; all-atom lipid RMSD below 2 Å.","version":null,"comparison":{"protocol_id":null,"dataset_version":"331 complexes","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-042","kind":"evaluation","name":"DiffDock-L: Lipid–protein binding pose","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"model","target_id":"reported-model-51ed86132346a0"},{"relation":"benchmark","target_id":"reported-task-ff2dec63c5a3dd"},{"relation":"dataset","target_id":"reported-dataset-6a44f5946cd7ab"}],"attributes":{"origin":"independent_paper","protocol":"Top-scoring pose; all-atom lipid RMSD below 2 Å.","version":null,"comparison":{"protocol_id":null,"dataset_version":"331 complexes","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-043","kind":"evaluation","name":"DiffDock-NMDN: Protein–ligand virtual screening","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"model","target_id":"reported-model-6c0bc8d297cc7a"},{"relation":"benchmark","target_id":"reported-task-a7803ecf7708cc"},{"relation":"dataset","target_id":"reported-dataset-065b9fcc8da573"}],"attributes":{"origin":"author_reported","protocol":"NMDN scoring on DiffDock-NMDN blind docked poses; not ligand-pose RMSD.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-044","kind":"evaluation","name":"Vina: Protein–ligand virtual screening","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"model","target_id":"reported-model-1e51ccbfd2de61"},{"relation":"benchmark","target_id":"reported-task-a7803ecf7708cc"},{"relation":"dataset","target_id":"reported-dataset-065b9fcc8da573"}],"attributes":{"origin":"independent_paper","protocol":"Vina scoring on the same DiffDock-NMDN blind docked poses; not ligand-pose RMSD.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-045","kind":"evaluation","name":"Boltz-1: Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"model","target_id":"reported-model-d9a06805b36b8a"},{"relation":"benchmark","target_id":"reported-task-bf513ed6db92c5"},{"relation":"dataset","target_id":"reported-dataset-5afaefb87c8a94"}],"attributes":{"origin":"independent_paper","protocol":"All entries; authors note this dataset contains structures seen during model training.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-046","kind":"evaluation","name":"DiffDock: Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"model","target_id":"reported-model-7f6ffd9e2a08be"},{"relation":"benchmark","target_id":"reported-task-bf513ed6db92c5"},{"relation":"dataset","target_id":"reported-dataset-5afaefb87c8a94"}],"attributes":{"origin":"independent_paper","protocol":"All entries; rigid-protein docking comparator; authors note this dataset contains structures seen during model training.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-047","kind":"evaluation","name":"Boltz-2: Ligand potency prediction using generated poses","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"model","target_id":"reported-model-cdc9aabf4efc04"},{"relation":"benchmark","target_id":"reported-task-d5f897ab0f6f67"},{"relation":"dataset","target_id":"reported-dataset-235520c84b737f"}],"attributes":{"origin":"independent_paper","protocol":"Potency prediction using Boltz-2 ligand-pose generation protocol; see paper scoring pipeline.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-048","kind":"evaluation","name":"DiffDock: Ligand potency prediction using generated poses","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"model","target_id":"reported-model-415ee22f46526c"},{"relation":"benchmark","target_id":"reported-task-d5f897ab0f6f67"},{"relation":"dataset","target_id":"reported-dataset-235520c84b737f"}],"attributes":{"origin":"independent_paper","protocol":"Potency prediction using DiffDock ligand-pose generation plus paper scoring pipeline; not a native DiffDock affinity score.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-003","kind":"evaluation","name":"Mouse-Geneformer: Human thymus cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"model","target_id":"reported-model-e2f2f0d4830bb0"},{"relation":"benchmark","target_id":"reported-task-031186b57c62de"},{"relation":"dataset","target_id":"reported-dataset-477a9082515406"}],"attributes":{"origin":"author_reported","protocol":"Ortholog-based gene conversion; zero-shot mouse model on human cells.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-004","kind":"evaluation","name":"Human-Geneformer: Human thymus cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"model","target_id":"reported-model-10d85f2a035720"},{"relation":"benchmark","target_id":"reported-task-031186b57c62de"},{"relation":"dataset","target_id":"reported-dataset-477a9082515406"}],"attributes":{"origin":"independent_paper","protocol":"Native human model; zero-shot setting.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-005","kind":"evaluation","name":"scLLMDA: Cross-platform scATAC cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"model","target_id":"reported-model-a0db32ae53e5ed"},{"relation":"benchmark","target_id":"reported-task-d82b6284f3f431"},{"relation":"dataset","target_id":"reported-dataset-7fc59ce4c0ceaa"}],"attributes":{"origin":"author_reported","protocol":"Cross-platform reference-query cell-type annotation.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-006","kind":"evaluation","name":"MINGLE: Cross-platform scATAC cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"model","target_id":"reported-model-f0c630d0565e64"},{"relation":"benchmark","target_id":"reported-task-d82b6284f3f431"},{"relation":"dataset","target_id":"reported-dataset-7fc59ce4c0ceaa"}],"attributes":{"origin":"independent_paper","protocol":"Cross-platform reference-query comparator.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-011","kind":"evaluation","name":"GenePT-w: Cell-type structure in frozen embeddings","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"model","target_id":"reported-model-7c595040de69bc"},{"relation":"benchmark","target_id":"reported-task-4df1fb456d3deb"},{"relation":"dataset","target_id":"reported-dataset-eaa2965545c87b"}],"attributes":{"origin":"author_reported","protocol":"k-means on pretrained cell embeddings; agreement with original cell-type labels.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-012","kind":"evaluation","name":"scGPT: Cell-type structure in frozen embeddings","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"model","target_id":"reported-model-399b1ce87a3f6d"},{"relation":"benchmark","target_id":"reported-task-4df1fb456d3deb"},{"relation":"dataset","target_id":"reported-dataset-eaa2965545c87b"}],"attributes":{"origin":"independent_paper","protocol":"k-means on pretrained cell embeddings; agreement with original cell-type labels.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-013","kind":"evaluation","name":"Best frozen single-cell foundation model: Donor-aware age-class prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"model","target_id":"reported-model-148b613975b6eb"},{"relation":"benchmark","target_id":"reported-task-d6018ca598e525"},{"relation":"dataset","target_id":"reported-dataset-571ce000cd74b9"}],"attributes":{"origin":"independent_paper","protocol":"Same donor-aware splits and logistic-regression probe as expression PCA; text names Geneformer as best model on AIDA v2.","version":null,"comparison":{"protocol_id":null,"dataset_version":"622 donors","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-014","kind":"evaluation","name":"Gene-expression PCA: Donor-aware age-class prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"model","target_id":"reported-model-177f32ce8189a0"},{"relation":"benchmark","target_id":"reported-task-d6018ca598e525"},{"relation":"dataset","target_id":"reported-dataset-571ce000cd74b9"}],"attributes":{"origin":"independent_paper","protocol":"Fifty-component gene-expression PCA with the same donor-aware probe splits.","version":null,"comparison":{"protocol_id":null,"dataset_version":"622 donors","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-015","kind":"evaluation","name":"scaLR: PBMC cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"model","target_id":"reported-model-7cf2f9951e1dba"},{"relation":"benchmark","target_id":"reported-task-b46b7b839bff93"},{"relation":"dataset","target_id":"reported-dataset-33a41fe5fc66cf"}],"attributes":{"origin":"author_reported","protocol":"All features and samples from PBMCs-BS.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-016","kind":"evaluation","name":"scVI + scANVI: PBMC cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"model","target_id":"reported-model-1be5c4b7c52a41"},{"relation":"benchmark","target_id":"reported-task-b46b7b839bff93"},{"relation":"dataset","target_id":"reported-dataset-33a41fe5fc66cf"}],"attributes":{"origin":"independent_paper","protocol":"All features and samples from PBMCs-BS; comparison pipeline combines scVI and scANVI.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-017","kind":"evaluation","name":"scXDR: Cross-dataset single-cell drug response transfer","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"model","target_id":"reported-model-9f39de53f7a139"},{"relation":"benchmark","target_id":"reported-task-167f08013c270e"},{"relation":"dataset","target_id":"reported-dataset-f7210686a78474"}],"attributes":{"origin":"author_reported","protocol":"Single-cell-to-single-cell transfer; source scenario 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-018","kind":"evaluation","name":"scVI: Cross-dataset single-cell drug response transfer","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"model","target_id":"reported-model-f23306b94dc7b6"},{"relation":"benchmark","target_id":"reported-task-167f08013c270e"},{"relation":"dataset","target_id":"reported-dataset-f7210686a78474"}],"attributes":{"origin":"independent_paper","protocol":"Single-cell-to-single-cell transfer; source scenario 2.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-019","kind":"evaluation","name":"CAMMiQ: Strain-level abundance quantification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"model","target_id":"reported-model-32a19f43a4c254"},{"relation":"benchmark","target_id":"reported-task-571f0a2e7faed3"},{"relation":"dataset","target_id":"reported-dataset-72e837e5b97041"}],"attributes":{"origin":"author_reported","protocol":"Strain-level quantification on the HumanGut-all synthetic query.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-020","kind":"evaluation","name":"Kraken2: Strain-level abundance quantification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"model","target_id":"reported-model-70f57ebb163a5c"},{"relation":"benchmark","target_id":"reported-task-571f0a2e7faed3"},{"relation":"dataset","target_id":"reported-dataset-72e837e5b97041"}],"attributes":{"origin":"independent_paper","protocol":"Strain-level quantification on the HumanGut-all synthetic query.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-021","kind":"evaluation","name":"Lazypipe-nt: Simulated metagenome virus-taxon retrieval","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"model","target_id":"reported-model-8cb3dd4e9f5b10"},{"relation":"benchmark","target_id":"reported-task-369dcfef14c4a9"},{"relation":"dataset","target_id":"reported-dataset-450c1af18cc623"}],"attributes":{"origin":"author_reported","protocol":"Genus-rank viral taxon retrieval.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-022","kind":"evaluation","name":"Kraken2: Simulated metagenome virus-taxon retrieval","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"model","target_id":"reported-model-673b8f46361000"},{"relation":"benchmark","target_id":"reported-task-369dcfef14c4a9"},{"relation":"dataset","target_id":"reported-dataset-450c1af18cc623"}],"attributes":{"origin":"independent_paper","protocol":"Genus-rank viral taxon retrieval.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-023","kind":"evaluation","name":"NCD-gzip: CAMI II superkingdom read classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["CAMI II superkingdom read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"model","target_id":"reported-model-7b052acf17b5ba"},{"relation":"benchmark","target_id":"reported-task-a2c37b8c420bc3"},{"relation":"dataset","target_id":"reported-dataset-beb4f5da29da0a"}],"attributes":{"origin":"author_reported","protocol":"Superkingdom-level macro-averaged F1; NCD assigns every read.","version":null,"comparison":{"protocol_id":null,"dataset_version":"10,000 reads","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-024","kind":"evaluation","name":"NCD-gzip: CAMI II phylum read classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["CAMI II phylum read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"model","target_id":"reported-model-7b052acf17b5ba"},{"relation":"benchmark","target_id":"reported-task-45105e1c486251"},{"relation":"dataset","target_id":"reported-dataset-beb4f5da29da0a"}],"attributes":{"origin":"author_reported","protocol":"Phylum-level macro-averaged F1; distinct taxonomic rank from the other row.","version":null,"comparison":{"protocol_id":null,"dataset_version":"10,000 reads","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-025","kind":"evaluation","name":"VIBRANT: Simulated prophage-contig detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"model","target_id":"reported-model-27dca28a87cf3c"},{"relation":"benchmark","target_id":"reported-task-53e3d216eef6db"},{"relation":"dataset","target_id":"reported-dataset-cd51026cdb6a7a"}],"attributes":{"origin":"independent_paper","protocol":"Average across twenty medium- and high-complexity simulated communities.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-026","kind":"evaluation","name":"VirSorter: Simulated prophage-contig detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"model","target_id":"reported-model-e78e3886df0d3a"},{"relation":"benchmark","target_id":"reported-task-53e3d216eef6db"},{"relation":"dataset","target_id":"reported-dataset-cd51026cdb6a7a"}],"attributes":{"origin":"independent_paper","protocol":"Average across twenty medium- and high-complexity simulated communities.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-027","kind":"evaluation","name":"GenomeOcean: Natural vs artificial microbial genome sequence","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"model","target_id":"reported-model-2df975e60d16d1"},{"relation":"benchmark","target_id":"reported-task-9f9ab0090f6522"},{"relation":"dataset","target_id":"reported-dataset-0c3ac7efe99c37"}],"attributes":{"origin":"author_reported","protocol":"Source reports natural-versus-artificial sequence classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-028","kind":"evaluation","name":"DNABERT-2: Natural vs artificial microbial genome sequence","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"model","target_id":"reported-model-7f1165b35f10e2"},{"relation":"benchmark","target_id":"reported-task-9f9ab0090f6522"},{"relation":"dataset","target_id":"reported-dataset-0c3ac7efe99c37"}],"attributes":{"origin":"independent_paper","protocol":"Source reports natural-versus-artificial sequence classification.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-029","kind":"evaluation","name":"kMetaShot: Mock-community MAG taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"model","target_id":"reported-model-790768ed581685"},{"relation":"benchmark","target_id":"reported-task-8406b6aabfb8c0"},{"relation":"dataset","target_id":"reported-dataset-8cad416ddc80dc"}],"attributes":{"origin":"author_reported","protocol":"Genus classification of MAGs from MegaHIT contigs; uncorrected kMetaShot.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-030","kind":"evaluation","name":"GTDB-Tk: Mock-community MAG taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"model","target_id":"reported-model-3fd1e9f6c573b2"},{"relation":"benchmark","target_id":"reported-task-8406b6aabfb8c0"},{"relation":"dataset","target_id":"reported-dataset-8cad416ddc80dc"}],"attributes":{"origin":"independent_paper","protocol":"Genus classification of the same MAG set.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-031","kind":"evaluation","name":"Lemur: Long-read taxonomic profiling","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"model","target_id":"reported-model-6ac0730e8481de"},{"relation":"benchmark","target_id":"reported-task-6330d593980b5b"},{"relation":"dataset","target_id":"reported-dataset-768a7ff5bac414"}],"attributes":{"origin":"author_reported","protocol":"Mean across five replicate runs on Zymo LOG 10%.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-032","kind":"evaluation","name":"Kraken 2: Long-read taxonomic profiling","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"model","target_id":"reported-model-62bc5e5ba13e7d"},{"relation":"benchmark","target_id":"reported-task-6330d593980b5b"},{"relation":"dataset","target_id":"reported-dataset-768a7ff5bac414"}],"attributes":{"origin":"independent_paper","protocol":"Mean across five replicate runs on Zymo LOG 10%.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-033","kind":"evaluation","name":"iPro-MP: Multi-species prokaryotic promoter detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"model","target_id":"reported-model-9a200c55b0e03e"},{"relation":"benchmark","target_id":"reported-task-a1151e386a3d3f"},{"relation":"dataset","target_id":"reported-dataset-e0f34dcaa1ba3b"}],"attributes":{"origin":"author_reported","protocol":"Average over independent testing sets.","version":null,"comparison":{"protocol_id":null,"dataset_version":"23 test sets","split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-034","kind":"evaluation","name":"Prompt: Multi-species prokaryotic promoter detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"model","target_id":"reported-model-03080a5289c07e"},{"relation":"benchmark","target_id":"reported-task-a1151e386a3d3f"},{"relation":"dataset","target_id":"reported-dataset-e0f34dcaa1ba3b"}],"attributes":{"origin":"independent_paper","protocol":"Average over the same independent testing sets.","version":null,"comparison":{"protocol_id":null,"dataset_version":"23 test sets","split":"independent test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-035","kind":"evaluation","name":"ICCTax: Hierarchical metagenomic taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"model","target_id":"reported-model-67eaf766fa9877"},{"relation":"benchmark","target_id":"reported-task-f4b1c9373f0929"},{"relation":"dataset","target_id":"reported-dataset-561834dfa1682c"}],"attributes":{"origin":"author_reported","protocol":"Macro average precision at genus rank on Complete dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-036","kind":"evaluation","name":"Kraken2: Hierarchical metagenomic taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"model","target_id":"reported-model-8861b9ad9b9c9b"},{"relation":"benchmark","target_id":"reported-task-f4b1c9373f0929"},{"relation":"dataset","target_id":"reported-dataset-561834dfa1682c"}],"attributes":{"origin":"independent_paper","protocol":"Macro average precision at genus rank on Complete dataset.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-037","kind":"evaluation","name":"Chai-1: Antibody–antigen interaction prediction using folded complexes","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"model","target_id":"reported-model-b3fdf259d51533"},{"relation":"benchmark","target_id":"reported-task-0c92cda11228c4"},{"relation":"dataset","target_id":"reported-dataset-ee26acbd6e8cf7"}],"attributes":{"origin":"independent_paper","protocol":"Interaction classifier evaluated using Chai-1-folded input complexes; this is pipeline AUC, not DockQ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-038","kind":"evaluation","name":"Boltz-1: Antibody–antigen interaction prediction using folded complexes","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"model","target_id":"reported-model-884582fb0c70dc"},{"relation":"benchmark","target_id":"reported-task-0c92cda11228c4"},{"relation":"dataset","target_id":"reported-dataset-ee26acbd6e8cf7"}],"attributes":{"origin":"independent_paper","protocol":"Interaction classifier evaluated using Boltz-1-folded input complexes; this is pipeline AUC, not DockQ.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-039","kind":"evaluation","name":"Boltz-1: Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz1-2025"],"links":[{"relation":"model","target_id":"reported-model-4058c43eb73b90"},{"relation":"benchmark","target_id":"reported-task-c04bb5ee6ecea6"},{"relation":"dataset","target_id":"reported-dataset-e45a5a140888ee"}],"attributes":{"origin":"author_reported","protocol":"Highest-confidence pose from five samples; precomputed MSAs up to 4,096 sequences.","version":"3 recycling rounds; 200 diffusion steps","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-040","kind":"evaluation","name":"Ibex: Antibody loop structure prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"model","target_id":"reported-model-2ae5fb0c147618"},{"relation":"benchmark","target_id":"reported-task-f3a12dbc0e0439"},{"relation":"dataset","target_id":"reported-dataset-b7204b005bd476"}],"attributes":{"origin":"author_reported","protocol":"Backbone RMSD after framework alignment; average over antibody test structures.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-041","kind":"evaluation","name":"Chai-1: Antibody loop structure prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"model","target_id":"reported-model-70c732770a200f"},{"relation":"benchmark","target_id":"reported-task-f3a12dbc0e0439"},{"relation":"dataset","target_id":"reported-dataset-b7204b005bd476"}],"attributes":{"origin":"independent_paper","protocol":"Backbone RMSD after framework alignment; one seed and one diffusion trajectory.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-042","kind":"evaluation","name":"DEELIG: Protein–ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"model","target_id":"reported-model-fa2da404b4d08e"},{"relation":"benchmark","target_id":"reported-task-d81be76396e644"},{"relation":"dataset","target_id":"reported-dataset-327cfcdae0c937"}],"attributes":{"origin":"author_reported","protocol":"Source paper reports DEELIG on PDBbind core set.","version":null,"comparison":{"protocol_id":null,"dataset_version":"v2016","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-043","kind":"evaluation","name":"TOPBP (Complex): Protein–ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"model","target_id":"reported-model-0068c3eff1bf7b"},{"relation":"benchmark","target_id":"reported-task-d81be76396e644"},{"relation":"dataset","target_id":"reported-dataset-327cfcdae0c937"}],"attributes":{"origin":"paper_compilation","protocol":"Source table compiles a previously published comparator; protocol equivalence is not established.","version":null,"comparison":{"protocol_id":null,"dataset_version":"v2016","split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","original_evaluation":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-044","kind":"evaluation","name":"MolAS: Physically valid protein–ligand pose selection","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"model","target_id":"reported-model-0eb4b0535b58e3"},{"relation":"benchmark","target_id":"reported-task-d1c46526c39983"},{"relation":"dataset","target_id":"reported-dataset-4b6c13924d4256"}],"attributes":{"origin":"author_reported","protocol":"Averaged five-fold algorithm-selection performance on PoseBusters; joint RMSD and validity criterion.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-045","kind":"evaluation","name":"Single best solver: Physically valid protein–ligand pose selection","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"model","target_id":"reported-model-028e4bb9baa074"},{"relation":"benchmark","target_id":"reported-task-d1c46526c39983"},{"relation":"dataset","target_id":"reported-dataset-4b6c13924d4256"}],"attributes":{"origin":"independent_paper","protocol":"Single best solver baseline under the same averaged five-fold selection test.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-046","kind":"evaluation","name":"AutoDock Vina holo: Intrinsically disordered protein ensemble docking","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"model","target_id":"reported-model-1587ab674d30a2"},{"relation":"benchmark","target_id":"reported-task-dec9e0f5e3da2a"},{"relation":"dataset","target_id":"reported-dataset-00201f65c32f6d"}],"attributes":{"origin":"independent_paper","protocol":"Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-047","kind":"evaluation","name":"DiffDock holo: Intrinsically disordered protein ensemble docking","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"model","target_id":"reported-model-75e4e5e5965320"},{"relation":"benchmark","target_id":"reported-task-dec9e0f5e3da2a"},{"relation":"dataset","target_id":"reported-dataset-00201f65c32f6d"}],"attributes":{"origin":"independent_paper","protocol":"Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-048","kind":"evaluation","name":"AK-score-ensemble: Protein–ligand binding affinity scoring","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"model","target_id":"reported-model-54be8a811c206e"},{"relation":"benchmark","target_id":"reported-task-a78312d5df6dad"},{"relation":"dataset","target_id":"reported-dataset-f18fcc23dfa798"}],"attributes":{"origin":"author_reported","protocol":"CASF-2016 scoring-power evaluation.","version":"ensemble; learning rate 0.0007","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-049","kind":"evaluation","name":"AK-score-single: Protein–ligand binding affinity scoring","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"model","target_id":"reported-model-de89576d8d316b"},{"relation":"benchmark","target_id":"reported-task-a78312d5df6dad"},{"relation":"dataset","target_id":"reported-dataset-f18fcc23dfa798"}],"attributes":{"origin":"author_reported","protocol":"CASF-2016 scoring-power evaluation.","version":"single; learning rate 0.0007","comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-050","kind":"evaluation","name":"PMF + ECFP + PF (LightGBM): Protein–ligand binding energy prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"model","target_id":"reported-model-3c196326586fa7"},{"relation":"benchmark","target_id":"reported-task-94802534b7026d"},{"relation":"dataset","target_id":"reported-dataset-16d01b5ef88e84"}],"attributes":{"origin":"author_reported","protocol":"Binding-energy model using ligand and protein fingerprints with LightGBM.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b3-051","kind":"evaluation","name":"PMF (LASSO): Protein–ligand binding energy prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"model","target_id":"reported-model-8100b3de6c7811"},{"relation":"benchmark","target_id":"reported-task-94802534b7026d"},{"relation":"dataset","target_id":"reported-dataset-16d01b5ef88e84"}],"attributes":{"origin":"author_reported","protocol":"PMF-only LASSO baseline evaluated by the same authors.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-001","kind":"evaluation","name":"ARSENAL+ChromBPNet: regulatory-variant scoring","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory-variant scoring"]},"source_ids":["arsenal-regulatory-dna-2026"],"links":[{"relation":"model","target_id":"reported-model-ee1ae8162c7d67"},{"relation":"benchmark","target_id":"reported-task-b9199a30a0bcb2"},{"relation":"dataset","target_id":"reported-dataset-158b121281b650"}],"attributes":{"origin":"author_reported","protocol":"Supervised ChromBPNet variant scoring with ARSENAL motif-discovery regularization","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-002","kind":"evaluation","name":"PlantCAD2: cross-species conservation prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["cross-species conservation prediction"]},"source_ids":["plantcad2-2025"],"links":[{"relation":"model","target_id":"reported-model-65059c3a806306"},{"relation":"benchmark","target_id":"reported-task-3109f8d0f2b7b5"},{"relation":"dataset","target_id":"reported-dataset-b6d0ebaca196a6"}],"attributes":{"origin":"author_reported","protocol":"Zero-shot score for conserved versus non-conserved sites from alignments of 35 Andropogoneae genomes","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-003","kind":"evaluation","name":"Stacking-Auto: enhancer prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["hi-enhancer-2025"],"links":[{"relation":"model","target_id":"reported-model-0829aff5471d4b"},{"relation":"benchmark","target_id":"reported-task-22024610c4d658"},{"relation":"dataset","target_id":"reported-dataset-a8610f2b80cdf0"}],"attributes":{"origin":"author_reported","protocol":"Two-stage Hi-Enhancer system; paper Table 2 method comparison","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-004","kind":"evaluation","name":"position-aware CNN: enhancer prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["enhancer-position-encoding-2024"],"links":[{"relation":"model","target_id":"reported-model-d2c81acf1c42c4"},{"relation":"benchmark","target_id":"reported-task-64607443a9ba15"},{"relation":"dataset","target_id":"reported-dataset-a03b8e9efde37b"}],"attributes":{"origin":"author_reported","protocol":"Nucleotide position-aware feature encoding; average assessment of CNN classifier","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-005","kind":"evaluation","name":"ADAR-GPT continual: A-to-I RNA editing site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["A-to-I RNA editing site prediction"]},"source_ids":["adar-gpt-editing-2026"],"links":[{"relation":"model","target_id":"reported-model-1d2aa9880a1c77"},{"relation":"benchmark","target_id":"reported-task-d635fc6c281a27"},{"relation":"dataset","target_id":"reported-dataset-236eaa4e55147f"}],"attributes":{"origin":"author_reported","protocol":"Curriculum plus 15% fine-tuning; 201-nt sequence windows; decision threshold 0.5","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"15% validation set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-006","kind":"evaluation","name":"R3Design: RNA sequence design","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA sequence design"]},"source_ids":["r3design-2025"],"links":[{"relation":"model","target_id":"reported-model-1b5fa066945d3d"},{"relation":"benchmark","target_id":"reported-task-df18c710f45213"},{"relation":"dataset","target_id":"reported-dataset-71614d99b3099f"}],"attributes":{"origin":"author_reported","protocol":"Tertiary-structure-conditioned RNA sequence design; external Rfam assessment","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"external","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-007","kind":"evaluation","name":"CUPID Data-aug-Avg: non-coding RNA pairwise interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["non-coding RNA pairwise interaction prediction"]},"source_ids":["cupid-rna-interactions-2026"],"links":[{"relation":"model","target_id":"reported-model-2894d253c5e8a8"},{"relation":"benchmark","target_id":"reported-task-c7ce06b753b8b6"},{"relation":"dataset","target_id":"reported-dataset-32ccef507a1dd7"}],"attributes":{"origin":"author_reported","protocol":"Data augmentation with average pooling for molecule-level ncRNA embeddings","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-008","kind":"evaluation","name":"ProteinBERT LLM-encoding model: mRNA-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA-protein interaction prediction"]},"source_ids":["mrna-protein-diversity-2026"],"links":[{"relation":"model","target_id":"reported-model-41ae49bb40ed8e"},{"relation":"benchmark","target_id":"reported-task-d7e6274011946e"},{"relation":"dataset","target_id":"reported-dataset-2e87449871ca47"}],"attributes":{"origin":"author_reported","protocol":"LLM encoding of protein partner; RBP-aware partition tests generalization to unseen protein diversity","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"RBP-aware test set","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-009","kind":"evaluation","name":"ESM2 650M: human-versus-viral protein classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["human-versus-viral protein classification"]},"source_ids":["viral-immune-mimicry-2025"],"links":[{"relation":"model","target_id":"reported-model-4c73500c39e9d0"},{"relation":"benchmark","target_id":"reported-task-53506fe386e4a1"},{"relation":"dataset","target_id":"reported-dataset-43f24c4dfb7351"}],"attributes":{"origin":"author_reported","protocol":"ESM2 650M embedding-based human-virus classifier","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-010","kind":"evaluation","name":"ProtT5 embeddings + ensemble classifier: protein-protein binding-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein binding-site prediction"]},"source_ids":["protein-binding-sites-2023"],"links":[{"relation":"model","target_id":"reported-model-fdac4c1ec8a433"},{"relation":"benchmark","target_id":"reported-task-f0ed5188dbb6d4"},{"relation":"dataset","target_id":"reported-dataset-f08b1a60aebeeb"}],"attributes":{"origin":"author_reported","protocol":"Explainable ensemble binding-site predictor using ProtT5 features","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-011","kind":"evaluation","name":"CLAPE-SMB with ESM-2: protein-small molecule binding-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-small molecule binding-site prediction"]},"source_ids":["clape-smb-2024"],"links":[{"relation":"model","target_id":"reported-model-57dbab30462150"},{"relation":"benchmark","target_id":"reported-task-b181ed450cdd41"},{"relation":"dataset","target_id":"reported-dataset-701d910b02d25c"}],"attributes":{"origin":"author_reported","protocol":"Contrastive CLAPE-SMB binding-site predictor with ESM-2 feature extractor","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-012","kind":"evaluation","name":"Vaxign-DL + ESM: vaccine-antigen candidate prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["vaccine-antigen candidate prediction"]},"source_ids":["vaxign-esm-2024"],"links":[{"relation":"model","target_id":"reported-model-8ad3e0cefde796"},{"relation":"benchmark","target_id":"reported-task-47465954d606e6"},{"relation":"dataset","target_id":"reported-dataset-b91c871eb7740a"}],"attributes":{"origin":"author_reported","protocol":"Combined skip architecture, four layers, ESM-generated sequence features","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-013","kind":"evaluation","name":"scGPT + residual geometry: gene-regulatory signal prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["gene-regulatory signal prediction"]},"source_ids":["single-cell-residual-geometry-2026"],"links":[{"relation":"model","target_id":"reported-model-d6a7fa854437e8"},{"relation":"benchmark","target_id":"reported-task-99afd88cb12895"},{"relation":"dataset","target_id":"reported-dataset-d9fdd8dc7a0184"}],"attributes":{"origin":"author_reported","protocol":"Asymmetric extraction, PCA-64 centered cosine geometry added to scGPT baseline","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-014","kind":"evaluation","name":"GREmLN: cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["cell-type annotation"]},"source_ids":["gremln-2026"],"links":[{"relation":"model","target_id":"reported-model-53d6515bcc1f39"},{"relation":"benchmark","target_id":"reported-task-6312c8a7ac045e"},{"relation":"dataset","target_id":"reported-dataset-59def895fbdbb4"}],"attributes":{"origin":"author_reported","protocol":"Zero-shot cell-type annotation using pre-trained cellular graph foundation model","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"zero-shot","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-015","kind":"evaluation","name":"Cell-DINO ViT-L: protein localization classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["protein localization classification"]},"source_ids":["cell-dino-2025"],"links":[{"relation":"model","target_id":"reported-model-808b23c65fbc89"},{"relation":"benchmark","target_id":"reported-task-7621fa1be55362"},{"relation":"dataset","target_id":"reported-dataset-d356eac961cb69"}],"attributes":{"origin":"author_reported","protocol":"Self-supervised microscopy embedding pre-trained on HPA-FoV; downstream protein-localization classifier. Dataset-specific pretraining; the paper does not claim a general-purpose foundation model that generalizes beyond these benchmarks.","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-016","kind":"evaluation","name":"scGen: differentially expressed gene identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["differentially expressed gene identification"]},"source_ids":["insilico-perturbation-auprc-2025"],"links":[{"relation":"model","target_id":"reported-model-fcf2cd29a81aae"},{"relation":"benchmark","target_id":"reported-task-003d746a129c9b"},{"relation":"dataset","target_id":"reported-dataset-7bf2cf7d2b2d01"}],"attributes":{"origin":"independent_paper","protocol":"In-silico perturbation assessment with precision sampled at fixed 50% recall","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"CD14+Mono","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-017","kind":"evaluation","name":"TCINet + HTRS: pathogen detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["pathogen detection"]},"source_ids":["metagenomic-pathogens-2025"],"links":[{"relation":"model","target_id":"reported-model-4ce8cae0f2eafc"},{"relation":"benchmark","target_id":"reported-task-d3fd502fdc2b38"},{"relation":"dataset","target_id":"reported-dataset-1c4c71078ffe01"}],"attributes":{"origin":"author_reported","protocol":"Taxonomy-constrained inference network with hierarchical taxonomy representation","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-018","kind":"evaluation","name":"DETIRE: viral sequence detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["viral sequence detection"]},"source_ids":["detire-viral-metagenomes-2023"],"links":[{"relation":"model","target_id":"reported-model-6d9dbac97852d8"},{"relation":"benchmark","target_id":"reported-task-3d4dec23120fef"},{"relation":"dataset","target_id":"reported-dataset-0b54f42a987b1d"}],"attributes":{"origin":"author_reported","protocol":"Hybrid deep learning virus-fragment classifier on paper testing dataset","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-019","kind":"evaluation","name":"PC-mer + LR: metagenomic genus classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["metagenomic genus classification"]},"source_ids":["pc-mer-2024"],"links":[{"relation":"model","target_id":"reported-model-688eb780ef7d2e"},{"relation":"benchmark","target_id":"reported-task-4420dcdfe8338d"},{"relation":"dataset","target_id":"reported-dataset-d28955d5872903"}],"attributes":{"origin":"author_reported","protocol":"k=8 PC-mer feature extraction with logistic regression on AMP genus-classification dataset","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"genus-level","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-020","kind":"evaluation","name":"MDL4Microbiome: microbiome disease-state classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["microbiome disease-state classification"]},"source_ids":["mdl4microbiome-2022"],"links":[{"relation":"model","target_id":"reported-model-e6ba198c2ac996"},{"relation":"benchmark","target_id":"reported-task-e2009c35eabd69"},{"relation":"dataset","target_id":"reported-dataset-bd3f98e2eeb5d3"}],"attributes":{"origin":"author_reported","protocol":"Multimodal deep learning model on colorectal-cancer versus healthy microbiome samples","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-021","kind":"evaluation","name":"binding-affinity meta-model: protein-ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["protein-ligand binding affinity prediction"]},"source_ids":["ligand-affinity-meta-model-2024"],"links":[{"relation":"model","target_id":"reported-model-df0efcc0346224"},{"relation":"benchmark","target_id":"reported-task-77a32496ce8fe6"},{"relation":"dataset","target_id":"reported-dataset-17132fbabd7683"}],"attributes":{"origin":"author_reported","protocol":"Sequence-or-structure meta-model; predicts ln(Kd/Ki) using docked and deep-learning components","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"core benchmark","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-022","kind":"evaluation","name":"DeepInterAware: antigen-antibody HIV neutralization prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["antigen-antibody HIV neutralization prediction"]},"source_ids":["deepinteraware-2025"],"links":[{"relation":"model","target_id":"reported-model-8d2c291733dfe1"},{"relation":"benchmark","target_id":"reported-task-45ead9a1eddf8d"},{"relation":"dataset","target_id":"reported-dataset-50f0bdb7cf9ca4"}],"attributes":{"origin":"author_reported","protocol":"Sequence-based interface-aware model, antibody-unseen split","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"antibody-unseen","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-023","kind":"evaluation","name":"TransBind: transcription-factor DNA binding-site prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["transcription-factor DNA binding-site prediction"]},"source_ids":["transbind-2026"],"links":[{"relation":"model","target_id":"reported-model-81b0394d5ac3e8"},{"relation":"benchmark","target_id":"reported-task-ac191e878dff5e"},{"relation":"dataset","target_id":"reported-dataset-034c60a2dabc73"}],"attributes":{"origin":"author_reported","protocol":"Integrates protein and DNA embeddings for TFBS prediction on paper test dataset","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":"test","population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"evaluation-lit-b4-024","kind":"evaluation","name":"ESM2_AMPS: protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["protein-protein interaction prediction"]},"source_ids":["esm2-amp-2025"],"links":[{"relation":"model","target_id":"reported-model-67ea6bd77b2ed1"},{"relation":"benchmark","target_id":"reported-task-09c3100b77dcc5"},{"relation":"dataset","target_id":"reported-dataset-9135087a16af1c"}],"attributes":{"origin":"author_reported","protocol":"ESM2-derived embeddings plus paper interaction predictor","version":null,"comparison":{"protocol_id":null,"dataset_version":null,"split":null,"population":null,"inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"fingerprint-scoring-2022","kind":"source","name":"Machine-Learning- and Knowledge-Based Scoring Functions Incorporating Ligand and Protein Fingerprints","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9178954/","version":"PMC archival version PMC9178954.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1021/acsomega.2c02822","publication_status":"peer_reviewed","year":2022,"artifact_sha256":"47bd60c6392b801095fdb604de06c0d4bda6f555bae58e9955e491a5abf60576","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9178954/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.439460+00:00","legacy_paper":{"id":"fingerprint-scoring-2022","title":"Machine-Learning- and Knowledge-Based Scoring Functions Incorporating Ligand and Protein Fingerprints","year":2022,"publication_status":"peer_reviewed","version":"PMC archival version PMC9178954.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9178954/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: ACS Omega; PMC ID: PMC9178954.","doi":"10.1021/acsomega.2c02822"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"fujisan-2024","kind":"source","name":"Enhanced prediction of protein functional identity through the integration of sequence and structural features","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11609699/","version":"PMC11609699.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1016/j.csbj.2024.11.028","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"db33e0542005ffae00cd644dfe697185b94c8823d5aee2768620a6db0c48e56f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11609699/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:35.728Z","legacy_paper":{"id":"fujisan-2024","title":"Enhanced prediction of protein functional identity through the integration of sequence and structural features","year":2024,"publication_status":"peer_reviewed","version":"PMC11609699.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11609699/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Computational and Structural Biotechnology Journal; PMC ID: PMC11609699.","doi":"10.1016/j.csbj.2024.11.028"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"fusion-breakpoint-foundation-models-2026","kind":"source","name":"Benchmarking genomic foundation models for binary classification of gene fusion breakpoints from DNA sequences","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13182013/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1186/s13040-026-00553-1","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"0f4d9de77f1e39cfd2164a20653d86370767da684dc22d17e09f589761abeb5f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13182013/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558209+00:00","legacy_paper":{"id":"fusion-breakpoint-foundation-models-2026","title":"Benchmarking genomic foundation models for binary classification of gene fusion breakpoints from DNA sequences","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13182013/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1186/s13040-026-00553-1","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: BioData Mining."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"genept-2024","kind":"source","name":"GenePT: A Simple But Effective Foundation Model for Genes and Cells Built From ChatGPT","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10614824/","version":"PMC archival version PMC10614824.2","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2023.10.16.562533","publication_status":"preprint","year":2024,"artifact_sha256":"230a2ec55458d9243eaeeebf3244df7409eb02d47f4b809ee56a06dcb6fdd047","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10614824/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.399274+00:00","legacy_paper":{"id":"genept-2024","title":"GenePT: A Simple But Effective Foundation Model for Genes and Cells Built From ChatGPT","year":2024,"publication_status":"preprint","version":"PMC archival version PMC10614824.2","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10614824/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC10614824.","doi":"10.1101/2023.10.16.562533"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"genomeocean-2025","kind":"source","name":"GenomeOcean: An Efficient Genome Foundation Model Trained on Large-Scale Metagenomic Assemblies","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838515/","version":"preprint archived 2025-02-05","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2025.01.30.635558","publication_status":"preprint","year":2025,"artifact_sha256":"3cc0df52522fccda23e3958f069c916b87ee50bb5c9a992fa37e25256546e145","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11838515/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.224Z","legacy_paper":{"id":"genomeocean-2025","title":"GenomeOcean: An Efficient Genome Foundation Model Trained on Large-Scale Metagenomic Assemblies","year":2025,"publication_status":"preprint","version":"preprint archived 2025-02-05","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838515/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11838515.","doi":"10.1101/2025.01.30.635558"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"genomic-tokenizer-selection-2025","kind":"source","name":"The impact of tokenizer selection in genomic language models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12453675/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf456","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"0a01c36fdd63f3f6db509777e61c3f87e8a298c810f8aef7974915aaa0655342","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12453675/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558210+00:00","legacy_paper":{"id":"genomic-tokenizer-selection-2025","title":"The impact of tokenizer selection in genomic language models","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12453675/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf456","notes":"Final Bioinformatics journal article Table 2, Caduceus (char) Regulatory MCC 0.778 checked directly; same study also has a bioRxiv manuscript."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"gremln-2026","kind":"source","name":"GREmLN: A Cellular Graph Structure Aware Transcriptomics Foundation Model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13060794/","version":"preprint version in PMC","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1101/2025.07.03.663009","publication_status":"preprint","year":2026,"artifact_sha256":"3a20c4ededb749fc3f1120baf16dcfebe3fcb30418a91c445cfd91a7b5fdf553","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13060794/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:57.502Z","legacy_paper":{"id":"gremln-2026","title":"GREmLN: A Cellular Graph Structure Aware Transcriptomics Foundation Model","year":2026,"publication_status":"preprint","version":"preprint version in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13060794/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: bioRxiv; PMC ID: PMC13060794. Preprint; table labels metric F1; paper does not specify macro in this row.","doi":"10.1101/2025.07.03.663009"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"gsmformer-ppi-2026","kind":"source","name":"Multimodal graph, surface, and language-based model for protein protein interaction prediction","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12873117/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-34758-x","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"9b364b5d73d16f2787f93f78f17dbe98b954ab9c2c64c1df960eec2e615eb3b4","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12873117/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558212+00:00","legacy_paper":{"id":"gsmformer-ppi-2026","title":"Multimodal graph, surface, and language-based model for protein protein interaction prediction","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12873117/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s41598-025-34758-x","notes":"Numeric result checked against Table 6 in primary full-text XML; journal/source: Scientific Reports."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"hi-enhancer-2025","kind":"source","name":"Hi-Enhancer: a two-stage framework for prediction and localization of enhancers based on Blending-KAN and Stacking-Auto models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12758598/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/bioinformatics/btaf441","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"c86488c9f60329b7a3c4370598e7a0a9e4c8c45d1758b87007bfc8242376b009","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12758598/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"hi-enhancer-2025","title":"Hi-Enhancer: a two-stage framework for prediction and localization of enhancers based on Blending-KAN and Stacking-Auto models","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12758598/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Bioinformatics; PMC ID: PMC12758598. Task-specific enhancer predictor; not a DNA foundation model. Comparison values from older papers excluded.","doi":"10.1093/bioinformatics/btaf441"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ibex-2025","kind":"source","name":"Conformation-aware structure prediction of antigen-recognizing immune proteins","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12710905/","version":"PMC archival version PMC12710905.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1080/19420862.2025.2602217","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"caa1109bd5fe7f6be703aa9d4afd6f4f1522bcbce6b7361650eb59618c2a9e14","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12710905/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.426811+00:00","legacy_paper":{"id":"ibex-2025","title":"Conformation-aware structure prediction of antigen-recognizing immune proteins","year":2025,"publication_status":"peer_reviewed","version":"PMC archival version PMC12710905.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12710905/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: mAbs; PMC ID: PMC12710905.","doi":"10.1080/19420862.2025.2602217"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"icctax-2025","kind":"source","name":"ICCTax: a hierarchical taxonomic classifier for metagenomic sequences on a large language model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12619997/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1093/bioadv/vbaf257","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"2ce0b48f1cde3aea7e561d92f4d7dc1525af7439ccd16f80bec0773e8812c8ec","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12619997/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.373Z","legacy_paper":{"id":"icctax-2025","title":"ICCTax: a hierarchical taxonomic classifier for metagenomic sequences on a large language model","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12619997/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Bioinformatics Advances; PMC ID: PMC12619997.","doi":"10.1093/bioadv/vbaf257"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"insilico-perturbation-auprc-2025","kind":"source","name":"AUPRC: a metric for evaluating the performance of in-silico perturbation methods in identifying differentially expressed genes","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12400816/","version":"PMC archival version PMC12400816.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/bib/bbaf426","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"2715709d94f84744afa32cafdcaa72efd206d63af8c60afe7619b2cb90108b6b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12400816/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"insilico-perturbation-auprc-2025","title":"AUPRC: a metric for evaluating the performance of in-silico perturbation methods in identifying differentially expressed genes","year":2025,"publication_status":"peer_reviewed","version":"PMC archival version PMC12400816.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12400816/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Briefings in Bioinformatics; PMC ID: PMC12400816. Paper benchmarks metrics and scGen perturbation method; no foundation-model result in this row.","doi":"10.1093/bib/bbaf426"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ipromp-2025","kind":"source","name":"iPro-MP: a BERT-based model to predict multiple prokaryotic promoters","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12516880/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1186/s13059-025-03819-9","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"d21541ee1f7a168da8e4a7c0f0e133c970cbe7bc41118f43a929f08b2fd2afd1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12516880/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:55.361Z","legacy_paper":{"id":"ipromp-2025","title":"iPro-MP: a BERT-based model to predict multiple prokaryotic promoters","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12516880/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Genome Biology; PMC ID: PMC12516880.","doi":"10.1186/s13059-025-03819-9"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"kmetashot-2025","kind":"source","name":"kMetaShot: a fast and reliable taxonomy classifier for metagenome-assembled genomes","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11695915/","version":"PMC archival version PMC11695915.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1093/bib/bbae680","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"4584e93ea035c1170b8756a0a52cbe99fe72e70bd09b5f1dee639ee104f78247","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11695915/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.417367+00:00","legacy_paper":{"id":"kmetashot-2025","title":"kMetaShot: a fast and reliable taxonomy classifier for metagenome-assembled genomes","year":2025,"publication_status":"peer_reviewed","version":"PMC archival version PMC11695915.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11695915/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Briefings in Bioinformatics; PMC ID: PMC11695915.","doi":"10.1093/bib/bbae680"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lambda-prophage-2026","kind":"source","name":"LAMBDA: A Prophage Detection Benchmark for Genomic Language Models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13041943/","version":"PMC13041943.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.64898/2026.03.26.714501","publication_status":"preprint","year":2026,"artifact_sha256":"22c2e218e87dce757907f6086a0e2ad37c13f785b34fff5bea7cfa1a6c276b16","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13041943/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:36.240Z","legacy_paper":{"id":"lambda-prophage-2026","title":"LAMBDA: A Prophage Detection Benchmark for Genomic Language Models","year":2026,"publication_status":"preprint","version":"PMC13041943.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13041943/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC13041943.","doi":"10.64898/2026.03.26.714501"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lazypipe-2020","kind":"source","name":"Novel NGS pipeline for virus discovery from a wide spectrum of hosts and sample types","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7772471/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1093/ve/veaa091","publication_status":"peer_reviewed","year":2020,"artifact_sha256":"77842d8e4f6b419e331ab5a01fdf8f9eb8604f259425d79602be896aad3d0ad1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7772471/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.412605+00:00","legacy_paper":{"id":"lazypipe-2020","title":"Novel NGS pipeline for virus discovery from a wide spectrum of hosts and sample types","year":2020,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7772471/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Virus Evolution; PMC ID: PMC7772471.","doi":"10.1093/ve/veaa091"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lemur-magnet-2024","kind":"source","name":"Lightweight taxonomic profiling of long-read metagenomic datasets with Lemur and Magnet","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11185576/","version":"PMC archival version PMC11185576.2","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1101/2024.06.01.596961","publication_status":"preprint","year":2024,"artifact_sha256":"4afb9195da447916eb6f733816e3640741c7ade08ea8920d205c3be7b3cce27a","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11185576/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.420493+00:00","legacy_paper":{"id":"lemur-magnet-2024","title":"Lightweight taxonomic profiling of long-read metagenomic datasets with Lemur and Magnet","year":2024,"publication_status":"preprint","version":"PMC archival version PMC11185576.2","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11185576/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: bioRxiv; PMC ID: PMC11185576.","doi":"10.1101/2024.06.01.596961"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ligand-affinity-meta-model-2024","kind":"source","name":"Improved Prediction of Ligand–Protein Binding Affinities by Meta-modeling","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11632770/","version":"PMC archival version PMC11632770.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1021/acs.jcim.4c01116","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"0be25fe75bc0b2eb8065136555763bbae5964ea3a8fdb8c5de79ff96445f6a29","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11632770/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"ligand-affinity-meta-model-2024","title":"Improved Prediction of Ligand–Protein Binding Affinities by Meta-modeling","year":2024,"publication_status":"peer_reviewed","version":"PMC archival version PMC11632770.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11632770/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC11632770. Mixed prediction units across Table 4 comparators; only meta-model PCC recorded.","doi":"10.1021/acs.jcim.4c01116"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lipp-2026","kind":"source","name":"The LiPP Benchmark Set for Modeling Lipid–Protein Complexes: Comparison of Co-Folding and Docking Methods","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13292216/","version":"PMC13292216.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acs.jcim.6c01457","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"6ff34f2f709a14858a3753abf9f8f6efa1e7e3c351f15c70cf64264193a9414e","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13292216/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.548973+00:00","legacy_paper":{"id":"lipp-2026","title":"The LiPP Benchmark Set for Modeling Lipid–Protein Complexes: Comparison of Co-Folding and Docking Methods","year":2026,"publication_status":"peer_reviewed","version":"PMC13292216.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13292216/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC13292216.","doi":"10.1021/acs.jcim.6c01457"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"lit-001","kind":"result","name":"Caduceus-Ph · AUC · Human 5mC","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-001"}],"attributes":{"printed_value":"0.783","numeric_value":"0.783","metric":"AUC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, Human 5mC row, Caduceus-Ph column; cell: 0.783","artifact_sha256":"5d8ca9bcf88cc1b38ad667906a2e4699b1aefa6d31c6f49259784930353f3202","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12663285/fullTextXML"},"legacy_id":"lit-001","legacy_row":{"id":"lit-001","paper_id":"dna-foundation-models-2025","domain_id":"dna-genomes","task":"Human 5mC detection","model":"Caduceus-Ph","model_version":"","dataset":"Human 5mC","dataset_version":"","split":"","metric":"AUC","value":"0.783","unit":"unitless","uncertainty":"","protocol":"Binary epigenetic-modification classification as reported in the paper.","source_locator":"Table 3, Human 5mC row, Caduceus-Ph column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-002","kind":"result","name":"NT-v2 · AUC · Human 5mC","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-002"}],"attributes":{"printed_value":"0.7377","numeric_value":"0.7377","metric":"AUC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Human 5mC row, NT-v2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.327Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, Human 5mC row, NT-v2 column; cell: 0.7377","artifact_sha256":"5d8ca9bcf88cc1b38ad667906a2e4699b1aefa6d31c6f49259784930353f3202","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12663285/fullTextXML"},"legacy_id":"lit-002","legacy_row":{"id":"lit-002","paper_id":"dna-foundation-models-2025","domain_id":"dna-genomes","task":"Human 5mC detection","model":"NT-v2","model_version":"","dataset":"Human 5mC","dataset_version":"","split":"","metric":"AUC","value":"0.7377","unit":"unitless","uncertainty":"","protocol":"Binary epigenetic-modification classification as reported in the paper.","source_locator":"Table 3, Human 5mC row, NT-v2 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12663285/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-003","kind":"result","name":"ENBED · Accuracy · Genomic Benchmarks Mouse Enhancers","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-003"}],"attributes":{"printed_value":"90.3","numeric_value":"90.3","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, Mouse Enhancers row, ENBED column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Mouse Enhancers row, ENBED column; cell: 90.3","artifact_sha256":"e95d4be70d32e61af5a92eda8ea66f25a2cc629e3f83e7b5241b13cde8bdb83b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/fullTextXML"},"legacy_id":"lit-003","legacy_row":{"id":"lit-003","paper_id":"enbed-2024","domain_id":"dna-genomes","task":"Enhancer classification","model":"ENBED","model_version":"","dataset":"Genomic Benchmarks Mouse Enhancers","dataset_version":"","split":"","metric":"Accuracy","value":"90.3","unit":"%","uncertainty":"","protocol":"Reported Genomic Benchmarks classification accuracy.","source_locator":"Table 2, Mouse Enhancers row, ENBED column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-004","kind":"result","name":"ENBED (GRCh38) · Accuracy · Genomic Benchmarks Mouse Enhancers","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-004"}],"attributes":{"printed_value":"81.1","numeric_value":"81.1","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.378Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column; cell: 81.1","artifact_sha256":"e95d4be70d32e61af5a92eda8ea66f25a2cc629e3f83e7b5241b13cde8bdb83b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11341122/fullTextXML"},"legacy_id":"lit-004","legacy_row":{"id":"lit-004","paper_id":"enbed-2024","domain_id":"dna-genomes","task":"Enhancer classification","model":"ENBED (GRCh38)","model_version":"","dataset":"Genomic Benchmarks Mouse Enhancers","dataset_version":"","split":"","metric":"Accuracy","value":"81.1","unit":"%","uncertainty":"","protocol":"ENBED trained on GRCh38; reported Genomic Benchmarks classification accuracy.","source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11341122/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-005","kind":"result","name":"DNABERT-2 · Accuracy · KEx","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-005"}],"attributes":{"printed_value":"97.0","numeric_value":"97.0","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":"± 0.5","source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, DNABERT-2 (117 M) row, Accuracy column; cell: 97.0 ± 0.5","artifact_sha256":"c3d7c6d068d3c11a9c8255a932197ece3804d94e8b2d4f858bea373a1b6eb32f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11953744/fullTextXML"},"legacy_id":"lit-005","legacy_row":{"id":"lit-005","paper_id":"quadruplex-llm-benchmark-2025","domain_id":"dna-genomes","task":"G-quadruplex classification","model":"DNABERT-2","model_version":"117M","dataset":"KEx","dataset_version":"","split":"","metric":"Accuracy","value":"97.0","unit":"%","uncertainty":"± 0.5","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11953744/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-006","kind":"result","name":"Caduceus · Accuracy · KEx","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-006"}],"attributes":{"printed_value":"95.0","numeric_value":"95.0","metric":"Accuracy","metric_direction":"unknown","unit":"%","uncertainty":"± 0.5","source_locator":"Table 5, Caduceus (8 M) row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.379Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, Caduceus (8 M) row, Accuracy column; cell: 95.0 ± 0.5","artifact_sha256":"c3d7c6d068d3c11a9c8255a932197ece3804d94e8b2d4f858bea373a1b6eb32f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11953744/fullTextXML"},"legacy_id":"lit-006","legacy_row":{"id":"lit-006","paper_id":"quadruplex-llm-benchmark-2025","domain_id":"dna-genomes","task":"G-quadruplex classification","model":"Caduceus","model_version":"8M","dataset":"KEx","dataset_version":"","split":"","metric":"Accuracy","value":"95.0","unit":"%","uncertainty":"± 0.5","protocol":"Pretrained model evaluated on KEx as reported in Table 5.","source_locator":"Table 5, Caduceus (8 M) row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11953744/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-007","kind":"result","name":"HyenaDNA · AUROC · DNALongBench ETGP","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-007"}],"attributes":{"printed_value":"0.828","numeric_value":"0.828","metric":"AUROC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, HyenaDNA row, ETGP column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.492545+00:00","notes":"ETGP is the first numeric column, separate from the six CMP cell-type columns and average; caption defines ETGP AUROC. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T3\", \"row_cells\": [\"HyenaDNA\", \"0.828\", \"0.139\", \"0.122\", \"0.099\", \"0.097\", \"0.118\", \"0.115\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"0.828\", \"caption\": \"AUROC for enhancer-target gene prediction (ETGP) task and SCC scores for the contact map prediction (CMP) task. K562, HFF, H1hESC, GM12878, IMR90, and HCT116 represent different human cell types. The highest scores are highlighted in bold. “Avg” means the average score across different cell types. Notably, the Expert Model achieves the best performance on both ETGP and CMP tasks.\"}","artifact_sha256":"fa440a17cecf16a5d872d50a30910f7591b5f6f78e10a944c6bda5ea8d7e32dd","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11741265/fullTextXML"},"legacy_id":"lit-007","legacy_row":{"id":"lit-007","paper_id":"dnalongbench-2025","domain_id":"dna-genomes","task":"Enhancer-target gene prediction","model":"HyenaDNA","model_version":"","dataset":"DNALongBench ETGP","dataset_version":"","split":"","metric":"AUROC","value":"0.828","unit":"unitless","uncertainty":"","protocol":"Long-range ETGP benchmark; source table reports AUROC.","source_locator":"Table 3, HyenaDNA row, ETGP column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11741265/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-008","kind":"result","name":"Caduceus-Ph · AUROC · DNALongBench ETGP","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-008"}],"attributes":{"printed_value":"0.826","numeric_value":"0.826","metric":"AUROC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Caduceus-Ph row, ETGP column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.493765+00:00","notes":"ETGP is the first numeric column, separate from the six CMP cell-type columns and average; caption defines ETGP AUROC. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T3\", \"row_cells\": [\"Caduceus-Ph\", \"0.826\", \"0.153\", \"0.130\", \"0.101\", \"0.138\", \"0.145\", \"0.133\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"0.826\", \"caption\": \"AUROC for enhancer-target gene prediction (ETGP) task and SCC scores for the contact map prediction (CMP) task. K562, HFF, H1hESC, GM12878, IMR90, and HCT116 represent different human cell types. The highest scores are highlighted in bold. “Avg” means the average score across different cell types. Notably, the Expert Model achieves the best performance on both ETGP and CMP tasks.\"}","artifact_sha256":"fa440a17cecf16a5d872d50a30910f7591b5f6f78e10a944c6bda5ea8d7e32dd","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11741265/fullTextXML"},"legacy_id":"lit-008","legacy_row":{"id":"lit-008","paper_id":"dnalongbench-2025","domain_id":"dna-genomes","task":"Enhancer-target gene prediction","model":"Caduceus-Ph","model_version":"","dataset":"DNALongBench ETGP","dataset_version":"","split":"","metric":"AUROC","value":"0.826","unit":"unitless","uncertainty":"","protocol":"Long-range ETGP benchmark; source table reports AUROC.","source_locator":"Table 3, Caduceus-Ph row, ETGP column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11741265/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-009","kind":"result","name":"RiNALMo · Pearson R · mRNABench MRL-MPRA","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-009"}],"attributes":{"printed_value":"0.74","numeric_value":"0.74","metric":"Pearson R","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, RiNALMo row, MRL MPRA column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.497221+00:00","notes":"MRL MPRA is the second task under Local and uses R. Caption specifies mean over ten random seeds and selected best model per family, not a fully identified checkpoint. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T9\", \"row_cells\": [\"RiNALMo\", \"31.6\", \"0.74\", \"39.2\", \"79.5\", \"69.0\", \"0.53\", \"0.42\", \"0.47\", \"32.7\", \"34.7\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"\\n\\n0.74\\n\\n\", \"caption\": \"Linear probe results. Mean of metric over ten random seeds reported. Best model per model family reported, see Appendix D for selected models. Best model for each dataset is and best foundation model is underlined. Models not significantly worse under Wilcoxon signed-rank test at p=0.05 are bolded.\"}","artifact_sha256":"79f6264ee883535203c63a313547e7c57baa85585f76b42f8d899eb17fb7e600","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12265608/fullTextXML"},"legacy_id":"lit-009","legacy_row":{"id":"lit-009","paper_id":"mrnabench-2025","domain_id":"rna-transcriptomes","task":"Mean ribosome load from MPRA","model":"RiNALMo","model_version":"","dataset":"mRNABench MRL-MPRA","dataset_version":"","split":"","metric":"Pearson R","value":"0.74","unit":"unitless","uncertainty":"","protocol":"Linear probe; mean across ten random seeds.","source_locator":"Table 2, RiNALMo row, MRL MPRA column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12265608/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-010","kind":"result","name":"RNA-FM · Pearson R · mRNABench MRL-MPRA","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-010"}],"attributes":{"printed_value":"0.49","numeric_value":"0.49","metric":"Pearson R","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, RNA-FM row, MRL MPRA column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.500211+00:00","notes":"MRL MPRA is the second task under Local and uses R. Caption specifies mean over ten random seeds and selected best model per family, not a fully identified checkpoint. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T9\", \"row_cells\": [\"RNA-FM\", \"25.7\", \"0.49\", \"35.0\", \"74.3\", \"67.0\", \"0.47\", \"0.29\", \"0.49\", \"32.2\", \"32.2\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"0.49\", \"caption\": \"Linear probe results. Mean of metric over ten random seeds reported. Best model per model family reported, see Appendix D for selected models. Best model for each dataset is and best foundation model is underlined. Models not significantly worse under Wilcoxon signed-rank test at p=0.05 are bolded.\"}","artifact_sha256":"79f6264ee883535203c63a313547e7c57baa85585f76b42f8d899eb17fb7e600","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12265608/fullTextXML"},"legacy_id":"lit-010","legacy_row":{"id":"lit-010","paper_id":"mrnabench-2025","domain_id":"rna-transcriptomes","task":"Mean ribosome load from MPRA","model":"RNA-FM","model_version":"","dataset":"mRNABench MRL-MPRA","dataset_version":"","split":"","metric":"Pearson R","value":"0.49","unit":"unitless","uncertainty":"","protocol":"Linear probe; mean across ten random seeds.","source_locator":"Table 2, RNA-FM row, MRL MPRA column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12265608/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-011","kind":"result","name":"BPfold · F1 · PDB RNA set","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-011"}],"attributes":{"printed_value":"0.814","numeric_value":"0.814","metric":"F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, BPfold row, PDB F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.502000+00:00","notes":"PDB is the second four-metric block; its F1 is numeric column six, not Rfam F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"Tab2\", \"row_cells\": [\"BPfold\", \"0.694\", \"0.689\", \"0.660\", \"0.741\", \"0.817\", \"0.814\", \"0.840\", \"0.801\"], \"selected_cell_zero_based\": 6, \"selected_cell_xml\": \"0.814\", \"caption\": \"Family-wise evaluation of three DL methods (BPfold, SPOT-RNA, and MXfold2), three shallow learning methods (ContextFold, CONTRAfold, and EternaFold) and non-learning methods (LinearFold, RNAfold, SimFold, and RNAstructure) on Rfam12.3–14.10 (n = 10,791 RNAs) and PDB (n = 116 RNAs) datasets\"}","artifact_sha256":"976218bd172998a1a6e7ed1609ecb8cb2ee380fb48a8dc7b25bc05ea8b0a49af","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12216785/fullTextXML"},"legacy_id":"lit-011","legacy_row":{"id":"lit-011","paper_id":"bpfold-2025","domain_id":"rna-transcriptomes","task":"RNA secondary structure","model":"BPfold","model_version":"","dataset":"PDB RNA set","dataset_version":"116 RNAs","split":"","metric":"F1","value":"0.814","unit":"unitless","uncertainty":"","protocol":"Family-wise evaluation of canonical base-pair predictions.","source_locator":"Table 2, BPfold row, PDB F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12216785/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-012","kind":"result","name":"RNAfold · F1 · PDB RNA set","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-012"}],"attributes":{"printed_value":"0.747","numeric_value":"0.747","metric":"F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, RNAfold row, PDB F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.504220+00:00","notes":"PDB is the second four-metric block; its F1 is numeric column six, not Rfam F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"Tab2\", \"row_cells\": [\"RNAfold\", \"0.656\", \"0.649\", \"0.599\", \"0.729\", \"0.749\", \"0.747\", \"0.776\", \"0.728\"], \"selected_cell_zero_based\": 6, \"selected_cell_xml\": \"0.747\", \"caption\": \"Family-wise evaluation of three DL methods (BPfold, SPOT-RNA, and MXfold2), three shallow learning methods (ContextFold, CONTRAfold, and EternaFold) and non-learning methods (LinearFold, RNAfold, SimFold, and RNAstructure) on Rfam12.3–14.10 (n = 10,791 RNAs) and PDB (n = 116 RNAs) datasets\"}","artifact_sha256":"976218bd172998a1a6e7ed1609ecb8cb2ee380fb48a8dc7b25bc05ea8b0a49af","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12216785/fullTextXML"},"legacy_id":"lit-012","legacy_row":{"id":"lit-012","paper_id":"bpfold-2025","domain_id":"rna-transcriptomes","task":"RNA secondary structure","model":"RNAfold","model_version":"","dataset":"PDB RNA set","dataset_version":"116 RNAs","split":"","metric":"F1","value":"0.747","unit":"unitless","uncertainty":"","protocol":"Family-wise evaluation of canonical base-pair predictions.","source_locator":"Table 2, RNAfold row, PDB F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12216785/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-013","kind":"result","name":"TU-Fold (aug) · F1 · RNA8F","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-013"}],"attributes":{"printed_value":"0.947","numeric_value":"0.947","metric":"F1","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.002 standard deviation","source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.505799+00:00","notes":"Overall is the first two-metric block. F1 is first numeric column; source cell includes uncertainty after the preserved central value. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl0020\", \"row_cells\": [\"TU-Fold (aug)\", \"0.947±0.002\", \"0.947±0.002\", \"0.973±0.003\", \"0.973±0.003\", \"0.797±0.016\", \"0.802±0.015\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"0.947±0.002\", \"caption\": \"Main results: the F1 score and the INF score of each method. The largest score in each column is highlighted in bold. The score with an underline is the second largest one in each column. The numbers on the left and right of the ‘±’ suggest the mean and the standard deviation of this score when training and evaluating the model using 3 folds of the dataset.\"}","artifact_sha256":"5aa376d6466daee83fc307baa39fd48da0f185ff30a178624025032d4cbe597d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12008525/fullTextXML"},"legacy_id":"lit-013","legacy_row":{"id":"lit-013","paper_id":"tu-fold-2025","domain_id":"rna-transcriptomes","task":"RNA secondary structure","model":"TU-Fold (aug)","model_version":"","dataset":"RNA8F","dataset_version":"","split":"","metric":"F1","value":"0.947","unit":"unitless","uncertainty":"± 0.002 standard deviation","protocol":"Three-fold training and evaluation; source reports mean and standard deviation.","source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12008525/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-014","kind":"result","name":"UFold · F1 · RNA8F","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-014"}],"attributes":{"printed_value":"0.938","numeric_value":"0.938","metric":"F1","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.004 standard deviation","source_locator":"Table 2, UFold row, Overall F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.507031+00:00","notes":"Overall is the first two-metric block. F1 is first numeric column; source cell includes uncertainty after the preserved central value. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl0020\", \"row_cells\": [\"UFold\", \"0.938±0.004\", \"0.938±0.004\", \"0.976±0.003\", \"0.976±0.003\", \"0.719±0.030\", \"0.720±0.030\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"0.938±0.004\", \"caption\": \"Main results: the F1 score and the INF score of each method. The largest score in each column is highlighted in bold. The score with an underline is the second largest one in each column. The numbers on the left and right of the ‘±’ suggest the mean and the standard deviation of this score when training and evaluating the model using 3 folds of the dataset.\"}","artifact_sha256":"5aa376d6466daee83fc307baa39fd48da0f185ff30a178624025032d4cbe597d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12008525/fullTextXML"},"legacy_id":"lit-014","legacy_row":{"id":"lit-014","paper_id":"tu-fold-2025","domain_id":"rna-transcriptomes","task":"RNA secondary structure","model":"UFold","model_version":"","dataset":"RNA8F","dataset_version":"","split":"","metric":"F1","value":"0.938","unit":"unitless","uncertainty":"± 0.004 standard deviation","protocol":"Three-fold training and evaluation; source reports mean and standard deviation.","source_locator":"Table 2, UFold row, Overall F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12008525/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-015","kind":"result","name":"DEBFold · Median F1 · DEBFold TestSetβ","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-015"}],"attributes":{"printed_value":"55.7","numeric_value":"55.7","metric":"Median F1","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.509290+00:00","notes":"TestSet beta is the second four-column block. F1 (%) is its first column; caption reports test-set median F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl1\", \"row_cells\": [\"DEBFold\", \"64.9\", \"62.1\", \"67.9\", \"4\", \"55.7\", \"56.4\", \"56.8\", \"1\", \"77.9\", \"83.3\", \"73.2\", \"1\", \"64.9\", \"1\"], \"selected_cell_zero_based\": 5, \"selected_cell_xml\": \"55.7\", \"caption\": \"Test Set Median F1 Score Performance Comparison between DEBFold and Other Available Thermodynamics-Based RNA Secondary Structure Prediction Tools on the Three Prepared Test Setsa\"}","artifact_sha256":"e8f960eafb7f00edfdd81d4fb75c6de838e9b872b7e18875fc7a5bff2a2f72b3","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11094721/fullTextXML"},"legacy_id":"lit-015","legacy_row":{"id":"lit-015","paper_id":"debfold-2024","domain_id":"rna-transcriptomes","task":"RNA secondary structure","model":"DEBFold","model_version":"","dataset":"DEBFold TestSetβ","dataset_version":"","split":"","metric":"Median F1","value":"55.7","unit":"%","uncertainty":"","protocol":"Median F1 on the prepared TestSetβ.","source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11094721/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-016","kind":"result","name":"RNAfold · Median F1 · DEBFold TestSetβ","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-016"}],"attributes":{"printed_value":"52.3","numeric_value":"52.3","metric":"Median F1","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 1, RNAfold row, TestSetβ F1 (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.511166+00:00","notes":"TestSet beta is the second four-column block. F1 (%) is its first column; caption reports test-set median F1. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl1\", \"row_cells\": [\"RNAfold\", \"63.5\", \"53.9\", \"82.9\", \"8\", \"52.3\", \"39.9\", \"81.2\", \"8\", \"77.9\", \"83.3\", \"73.2\", \"1\", \"63.5\", \"8\"], \"selected_cell_zero_based\": 5, \"selected_cell_xml\": \"52.3\", \"caption\": \"Test Set Median F1 Score Performance Comparison between DEBFold and Other Available Thermodynamics-Based RNA Secondary Structure Prediction Tools on the Three Prepared Test Setsa\"}","artifact_sha256":"e8f960eafb7f00edfdd81d4fb75c6de838e9b872b7e18875fc7a5bff2a2f72b3","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11094721/fullTextXML"},"legacy_id":"lit-016","legacy_row":{"id":"lit-016","paper_id":"debfold-2024","domain_id":"rna-transcriptomes","task":"RNA secondary structure","model":"RNAfold","model_version":"","dataset":"DEBFold TestSetβ","dataset_version":"","split":"","metric":"Median F1","value":"52.3","unit":"%","uncertainty":"","protocol":"Median F1 on the prepared TestSetβ.","source_locator":"Table 1, RNAfold row, TestSetβ F1 (%) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11094721/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-017","kind":"result","name":"ESM-2 · Mean Spearman rho · ProteinGym substitution DMS: stability assays","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-017"}],"attributes":{"printed_value":"0.488","numeric_value":"0.488","metric":"Mean Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table A7, ESM-2 (15B) row, Stability column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.517323+00:00","notes":"Table A7 is zero-shot substitution DMS grouped by function. Stability is the fifth numeric column; model-type row spans do not change its placement. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T11\", \"row_cells\": [\"ESM-2 (15B)\", \"0.405\", \"0.318\", \"0.425\", \"0.388\", \"0.488\", \"0.405\"], \"selected_cell_zero_based\": 5, \"selected_cell_xml\": \"0.488\", \"caption\": \"ProteinGym - Zero-shot substitution DMS benchmark by function typeAverage Spearman’s rank correlation between model scores and experimental measurements on the ProteinGym substitution benchmark, separated into five functional categories (Activity, Binding, Organismal Fitness, Stability and Expression). ‘All’ is the average of all the categories.\"}","artifact_sha256":"4519641f13271bdd09b166e7d93232f22542489bc52a25b5a1628c3df8badce1","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10723403/fullTextXML"},"legacy_id":"lit-017","legacy_row":{"id":"lit-017","paper_id":"proteingym-2023","domain_id":"proteins-complexes","task":"Zero-shot substitution mutation effects: stability","model":"ESM-2","model_version":"15B","dataset":"ProteinGym substitution DMS: stability assays","dataset_version":"","split":"","metric":"Mean Spearman rho","value":"0.488","unit":"unitless","uncertainty":"","protocol":"Zero-shot mutation scores; average Spearman across stability-category assays.","source_locator":"Table A7, ESM-2 (15B) row, Stability column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10723403/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-018","kind":"result","name":"ProteinMPNN · Mean Spearman rho · ProteinGym substitution DMS: stability assays","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-018"}],"attributes":{"printed_value":"0.566","numeric_value":"0.566","metric":"Mean Spearman rho","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table A7, ProteinMPNN row, Stability column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.523422+00:00","notes":"Table A7 is zero-shot substitution DMS grouped by function. Stability is the fifth numeric column; model-type row spans do not change its placement. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T11\", \"row_cells\": [\"ProteinMPNN\", \"0.197\", \"0.165\", \"0.198\", \"0.165\", \"0.566\", \"0.258\"], \"selected_cell_zero_based\": 5, \"selected_cell_xml\": \"0.566\", \"caption\": \"ProteinGym - Zero-shot substitution DMS benchmark by function typeAverage Spearman’s rank correlation between model scores and experimental measurements on the ProteinGym substitution benchmark, separated into five functional categories (Activity, Binding, Organismal Fitness, Stability and Expression). ‘All’ is the average of all the categories.\"}","artifact_sha256":"4519641f13271bdd09b166e7d93232f22542489bc52a25b5a1628c3df8badce1","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10723403/fullTextXML"},"legacy_id":"lit-018","legacy_row":{"id":"lit-018","paper_id":"proteingym-2023","domain_id":"proteins-complexes","task":"Zero-shot substitution mutation effects: stability","model":"ProteinMPNN","model_version":"","dataset":"ProteinGym substitution DMS: stability assays","dataset_version":"","split":"","metric":"Mean Spearman rho","value":"0.566","unit":"unitless","uncertainty":"","protocol":"Zero-shot mutation scores; average Spearman across stability-category assays.","source_locator":"Table A7, ProteinMPNN row, Stability column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10723403/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-019","kind":"result","name":"FUJISAN · AUROC · FUJISAN test sub-dataset","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-019"}],"attributes":{"printed_value":"0.9427","numeric_value":"0.9427","metric":"AUROC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, FUJISAN row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.728Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 1, FUJISAN row, AUROC column; cell: 0.9427","artifact_sha256":"db33e0542005ffae00cd644dfe697185b94c8823d5aee2768620a6db0c48e56f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11609699/fullTextXML"},"legacy_id":"lit-019","legacy_row":{"id":"lit-019","paper_id":"fujisan-2024","domain_id":"proteins-complexes","task":"Enzyme functional identity prediction","model":"FUJISAN","model_version":"","dataset":"FUJISAN test sub-dataset","dataset_version":"","split":"","metric":"AUROC","value":"0.9427","unit":"unitless","uncertainty":"","protocol":"Sequence and structural feature integration; paper-reported test sub-dataset.","source_locator":"Table 1, FUJISAN row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11609699/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-020","kind":"result","name":"ESM2 · AUROC · FUJISAN test sub-dataset","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-020"}],"attributes":{"printed_value":"0.7991","numeric_value":"0.7991","metric":"AUROC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, ESM2 row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:35.728Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 1, ESM2 row, AUROC column; cell: 0.7991","artifact_sha256":"db33e0542005ffae00cd644dfe697185b94c8823d5aee2768620a6db0c48e56f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11609699/fullTextXML"},"legacy_id":"lit-020","legacy_row":{"id":"lit-020","paper_id":"fujisan-2024","domain_id":"proteins-complexes","task":"Enzyme functional identity prediction","model":"ESM2","model_version":"","dataset":"FUJISAN test sub-dataset","dataset_version":"","split":"","metric":"AUROC","value":"0.7991","unit":"unitless","uncertainty":"","protocol":"Comparator evaluated on the paper-reported test sub-dataset.","source_locator":"Table 1, ESM2 row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11609699/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-021","kind":"result","name":"ESM-2 · R² · PRIME mutated RBD","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-021"}],"attributes":{"printed_value":"0.0248","numeric_value":"0.0248","metric":"R²","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.01","source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.525183+00:00","notes":"Resolved model row spans and Mean/CLS subrows in JATS: selected Mean, not fine-tuned (cross), Position-Stratified Split > Binding > R-squared. Central value agrees; uncertainty is retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"Tab1\", \"row_cells\": [\"Mean\", \"×\", \"0.6794 ± 0.02\", \"1.0777 ± 0.04\", \"0.6576 ± 0.04\", \"0.5807 ± 0.03\", \"0.0248 ± 0.01\", \"1.7519 ± 0.01\", \"0.0967 ± 0.02\", \"1.0221 ± 0.01\"], \"selected_cell_zero_based\": 6, \"selected_cell_xml\": \"0.0248 ± 0.01\", \"caption\": \"Benchmarking PRIME across different model scales and validation regimes for mutated RBD binding and expression\"}","artifact_sha256":"f6aac4c25dd93026f87ce9a2e327c95faf4c3014d7f9ae04bb11f208ce047971","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13425921/fullTextXML"},"legacy_id":"lit-021","legacy_row":{"id":"lit-021","paper_id":"prime-2026","domain_id":"proteins-complexes","task":"Mutated RBD binding prediction","model":"ESM-2","model_version":"8M","dataset":"PRIME mutated RBD","dataset_version":"","split":"position-stratified","metric":"R²","value":"0.0248","unit":"unitless","uncertainty":"± 0.01","protocol":"Frozen mean-pooled representation with downstream regression; position-stratified split.","source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13425921/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"lit-022","kind":"result","name":"ESM-C · R² · PRIME mutated RBD","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-022"}],"attributes":{"printed_value":"-0.0162","numeric_value":"-0.0162","metric":"R²","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.01","source_locator":"Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.526387+00:00","notes":"Resolved model row spans and Mean/CLS subrows in JATS: selected Mean, not fine-tuned (cross), Position-Stratified Split > Binding > R-squared. Central value agrees; uncertainty is retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"Tab1\", \"row_cells\": [\"Mean\", \"×\", \"0.7155 ± 0.07\", \"1.0075 ± 0.13\", \"0.6546 ± 0.04\", \"0.5830 ± 0.03\", \"-0.0162 ± 0.01\", \"1.7874 ± 0.02\", \"0.0647 ± 0.02\", \"1.0410 ± 0.01\"], \"selected_cell_zero_based\": 6, \"selected_cell_xml\": \"-0.0162 ± 0.01\", \"caption\": \"Benchmarking PRIME across different model scales and validation regimes for mutated RBD binding and expression\"}","artifact_sha256":"f6aac4c25dd93026f87ce9a2e327c95faf4c3014d7f9ae04bb11f208ce047971","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13425921/fullTextXML"},"legacy_id":"lit-022","legacy_row":{"id":"lit-022","paper_id":"prime-2026","domain_id":"proteins-complexes","task":"Mutated RBD binding prediction","model":"ESM-C","model_version":"300M","dataset":"PRIME mutated RBD","dataset_version":"","split":"position-stratified","metric":"R²","value":"-0.0162","unit":"unitless","uncertainty":"± 0.01","protocol":"Frozen mean-pooled representation with downstream regression; position-stratified split.","source_locator":"Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13425921/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract"}}} {"id":"lit-023","kind":"result","name":"PST · Mean |Spearman rho| · ProteinShake VEP datasets","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-023"}],"attributes":{"printed_value":"0.501","numeric_value":"0.501","metric":"Mean |Spearman rho|","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.527633+00:00","notes":"Zero-shot VEP is the last metric group; selected Mean absolute rho, not GO/EC/binding-site metrics. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"btaf582-T2\", \"row_cells\": [\"PST\", \"0.650\", \"0.883\", \"0.704\", \"0.436\", \"0.797\", \"0.501\"], \"selected_cell_zero_based\": 6, \"selected_cell_xml\": \"0.501\", \"caption\": \"Comparison of PST and ESM-2 on ProteinShake tasks and VEP datasets.a\"}","artifact_sha256":"c21ad593de589a7188ca86a8b7ce617e301d48dd939ecc7da03efb22cbe8d7a3","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12603367/fullTextXML"},"legacy_id":"lit-023","legacy_row":{"id":"lit-023","paper_id":"pst-2025","domain_id":"proteins-complexes","task":"Zero-shot variant effect prediction","model":"PST","model_version":"","dataset":"ProteinShake VEP datasets","dataset_version":"","split":"","metric":"Mean |Spearman rho|","value":"0.501","unit":"unitless","uncertainty":"","protocol":"Zero-shot VEP; paper averages absolute Spearman correlations.","source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12603367/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-024","kind":"result","name":"ESM-2 · Mean |Spearman rho| · ProteinShake VEP datasets","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-024"}],"attributes":{"printed_value":"0.489","numeric_value":"0.489","metric":"Mean |Spearman rho|","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.528656+00:00","notes":"Zero-shot VEP is the last metric group; selected Mean absolute rho, not GO/EC/binding-site metrics. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"btaf582-T2\", \"row_cells\": [\"ESM-2\", \"0.648\", \"0.858\", \"0.698\", \"0.431\", \"0.791\", \"0.489\"], \"selected_cell_zero_based\": 6, \"selected_cell_xml\": \"0.489\", \"caption\": \"Comparison of PST and ESM-2 on ProteinShake tasks and VEP datasets.a\"}","artifact_sha256":"c21ad593de589a7188ca86a8b7ce617e301d48dd939ecc7da03efb22cbe8d7a3","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12603367/fullTextXML"},"legacy_id":"lit-024","legacy_row":{"id":"lit-024","paper_id":"pst-2025","domain_id":"proteins-complexes","task":"Zero-shot variant effect prediction","model":"ESM-2","model_version":"","dataset":"ProteinShake VEP datasets","dataset_version":"","split":"","metric":"Mean |Spearman rho|","value":"0.489","unit":"unitless","uncertainty":"","protocol":"Zero-shot VEP; paper averages absolute Spearman correlations.","source_locator":"Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12603367/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-025","kind":"result","name":"scGPT · F1-Score · M.S. single-cell dataset","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-025"}],"attributes":{"printed_value":"0.734","numeric_value":"0.734","metric":"F1-Score","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, M.S. / scGPT row, F1-Score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.530269+00:00","notes":"Selected M.S. dataset block, first scGPT/Geneformer occurrences. F1-Score is last column; later dataset blocks deliberately excluded. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T2\", \"row_cells\": [\"M.S.\", \"scGPT\", \"0.595\", \"0.777\", \"0.728\", \"0.734\"], \"selected_cell_zero_based\": 5, \"selected_cell_xml\": \"0.734\", \"caption\": \"Performance of cell type identification using native scLLMs and popular tools.Bold value represents the highest score among the methods\"}","artifact_sha256":"77a4a859010259eadf2187465db6ab385efa4927a5eadb95c1e01991044c283f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10862733/fullTextXML"},"legacy_id":"lit-025","legacy_row":{"id":"lit-025","paper_id":"single-cell-peft-2024","domain_id":"cells-tissues","task":"Cell-type identification","model":"scGPT","model_version":"","dataset":"M.S. single-cell dataset","dataset_version":"","split":"","metric":"F1-Score","value":"0.734","unit":"unitless","uncertainty":"","protocol":"Native scLLM cell-type identification as reported in Table 2.","source_locator":"Table 2, M.S. / scGPT row, F1-Score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10862733/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-026","kind":"result","name":"Geneformer · F1-Score · M.S. single-cell dataset","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-026"}],"attributes":{"printed_value":"0.388","numeric_value":"0.388","metric":"F1-Score","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, M.S. / Geneformer row, F1-Score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.531756+00:00","notes":"Selected M.S. dataset block, first scGPT/Geneformer occurrences. F1-Score is last column; later dataset blocks deliberately excluded. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T2\", \"row_cells\": [\"\", \"Geneformer\", \"0.283\", \"0.235\", \"0.532\", \"0.388\"], \"selected_cell_zero_based\": 5, \"selected_cell_xml\": \"0.388\", \"caption\": \"Performance of cell type identification using native scLLMs and popular tools.Bold value represents the highest score among the methods\"}","artifact_sha256":"77a4a859010259eadf2187465db6ab385efa4927a5eadb95c1e01991044c283f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10862733/fullTextXML"},"legacy_id":"lit-026","legacy_row":{"id":"lit-026","paper_id":"single-cell-peft-2024","domain_id":"cells-tissues","task":"Cell-type identification","model":"Geneformer","model_version":"","dataset":"M.S. single-cell dataset","dataset_version":"","split":"","metric":"F1-Score","value":"0.388","unit":"unitless","uncertainty":"","protocol":"Native scLLM cell-type identification as reported in Table 2.","source_locator":"Table 2, M.S. / Geneformer row, F1-Score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10862733/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-027","kind":"result","name":"C2S (GPT-2 Large) · Partial-label accuracy · L1000","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-027"}],"attributes":{"printed_value":"0.631","numeric_value":"0.631","metric":"Partial-label accuracy","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.0031","source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.533640+00:00","notes":"Read inline small-caps/bold XML in document order, restoring Geneformer and GPT-2 Large labels. Selected Partial label (first block), L1000 > Acc, not AUROC or Full label. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T3\", \"row_cells\": [\"C2S (GPT-2 Large)\", \"0.639 ± 0.0049\", \"0.767 ± 0.0049\", \"0.631 ± 0.0031\", \"0.768 ± 0.0021\", \"0.575 ± 0.0035\", \"0.713 ± 0.0014\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"0.631 ± 0.0031\", \"caption\": \"Experimental results on downstream cell label classification. Cell labels are composed of multiple combinatorial metadata parts, including cell type, perturbations, and dosage information. Accuracy and area under ROC curve is computed on model predictions versus ground truth combinatorial labels, with partial credit given for partial misclassifications.\"}","artifact_sha256":"e088727d6e04857fccb7033a9b074e1850f775e86e7d2e99e603dde09558ab02","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11565894/fullTextXML"},"legacy_id":"lit-027","legacy_row":{"id":"lit-027","paper_id":"cell2sentence-2024","domain_id":"cells-tissues","task":"Combinatorial cell-label classification","model":"C2S (GPT-2 Large)","model_version":"GPT-2 Large","dataset":"L1000","dataset_version":"","split":"","metric":"Partial-label accuracy","value":"0.631","unit":"unitless","uncertainty":"± 0.0031","protocol":"Partial-credit labels including cell type, perturbation, and dose.","source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11565894/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-028","kind":"result","name":"Geneformer · Partial-label accuracy · L1000","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-028"}],"attributes":{"printed_value":"0.419","numeric_value":"0.419","metric":"Partial-label accuracy","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.0153","source_locator":"Table 3, Partial label / Geneformer row, L1000 Acc column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.535220+00:00","notes":"Read inline small-caps/bold XML in document order, restoring Geneformer and GPT-2 Large labels. Selected Partial label (first block), L1000 > Acc, not AUROC or Full label. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T3\", \"row_cells\": [\"Geneformer\", \"0.600 ± 0.0170\", \"0.722 ± 0.0145\", \"0.419 ± 0.0153\", \"0.632 ± 0.0181\", \"0.500 ± 0.0013\", \"0.649 ± 0.0025\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"0.419 ± 0.0153\", \"caption\": \"Experimental results on downstream cell label classification. Cell labels are composed of multiple combinatorial metadata parts, including cell type, perturbations, and dosage information. Accuracy and area under ROC curve is computed on model predictions versus ground truth combinatorial labels, with partial credit given for partial misclassifications.\"}","artifact_sha256":"e088727d6e04857fccb7033a9b074e1850f775e86e7d2e99e603dde09558ab02","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11565894/fullTextXML"},"legacy_id":"lit-028","legacy_row":{"id":"lit-028","paper_id":"cell2sentence-2024","domain_id":"cells-tissues","task":"Combinatorial cell-label classification","model":"Geneformer","model_version":"","dataset":"L1000","dataset_version":"","split":"","metric":"Partial-label accuracy","value":"0.419","unit":"unitless","uncertainty":"± 0.0153","protocol":"Partial-credit labels including cell type, perturbation, and dose.","source_locator":"Table 3, Partial label / Geneformer row, L1000 Acc column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11565894/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-029","kind":"result","name":"scGPT · F1 · hPancreas","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-029"}],"attributes":{"printed_value":"0.550","numeric_value":"0.550","metric":"F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.537541+00:00","notes":"Selected first hPancreas zero-shot block and F1 last column. Caption says some scores are copied from GenePT; this is source checking of the reported table, not independent experimental evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T1\", \"row_cells\": [\"scGPT (z)\", \"0.770\", \"0.610\", \"0.560\", \"0.550\"], \"selected_cell_zero_based\": 4, \"selected_cell_xml\": \"0.550\", \"caption\": \"Scores of cell-type annotation task under different settings. Parts of the results are directly extracted from GenePT. Here PCA represents principal component analysis, and scELMo+random emb represents fine-tuning scELMo with random numbers as meaningless gene embeddings. Average ranks of all methods across datasets are summarized in Extended Data Figure 6 (b). We boldfaced the highest score of each metric for each dataset.\"}","artifact_sha256":"ef75f0d63a567f5e9d7132fd847f44838a82a9741ae55323437e1d1812d86316","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12393277/fullTextXML"},"legacy_id":"lit-029","legacy_row":{"id":"lit-029","paper_id":"scelmo-2025","domain_id":"cells-tissues","task":"Cell-type annotation","model":"scGPT","model_version":"","dataset":"hPancreas","dataset_version":"","split":"","metric":"F1","value":"0.550","unit":"unitless","uncertainty":"","protocol":"Zero-shot setting; source caption says some comparator rows come from GenePT.","source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12393277/","evaluation_origin":"paper_compilation","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-030","kind":"result","name":"Geneformer · F1 · hPancreas","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-030"}],"attributes":{"printed_value":"0.270","numeric_value":"0.270","metric":"F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.539630+00:00","notes":"Selected first hPancreas zero-shot block and F1 last column. Caption says some scores are copied from GenePT; this is source checking of the reported table, not independent experimental evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T1\", \"row_cells\": [\"Geneformer (z)\", \"0.500\", \"0.250\", \"0.340\", \"0.270\"], \"selected_cell_zero_based\": 4, \"selected_cell_xml\": \"0.270\", \"caption\": \"Scores of cell-type annotation task under different settings. Parts of the results are directly extracted from GenePT. Here PCA represents principal component analysis, and scELMo+random emb represents fine-tuning scELMo with random numbers as meaningless gene embeddings. Average ranks of all methods across datasets are summarized in Extended Data Figure 6 (b). We boldfaced the highest score of each metric for each dataset.\"}","artifact_sha256":"ef75f0d63a567f5e9d7132fd847f44838a82a9741ae55323437e1d1812d86316","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12393277/fullTextXML"},"legacy_id":"lit-030","legacy_row":{"id":"lit-030","paper_id":"scelmo-2025","domain_id":"cells-tissues","task":"Cell-type annotation","model":"Geneformer","model_version":"","dataset":"hPancreas","dataset_version":"","split":"","metric":"F1","value":"0.270","unit":"unitless","uncertainty":"","protocol":"Zero-shot setting; source caption says some comparator rows come from GenePT.","source_locator":"Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12393277/","evaluation_origin":"paper_compilation","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-031","kind":"result","name":"scRegNet (Geneformer backbone) · AUROC · hESC cell-type-specific GRN","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-031"}],"attributes":{"printed_value":"0.89","numeric_value":"0.89","metric":"AUROC","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.00 as printed","source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.541287+00:00","notes":"Read break elements: cells contain AUROC on first line then AUPRC. Selected hESC (first cell type), first line. Caption specifies 500 most-variable genes and 50 independent evaluations. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T2\", \"row_cells\": [\"scRegNet (w/ Geneformer)\", \"AUROCAUPRC\", \"0.89±0.000.62±0.00\", \"0.90±0.000.84±0.00\", \"0.81±0.000.17±0.00\", \"0.93±0.000.86±0.00\", \"0.92±0.000.94±0.00\", \"0.93±0.000.94±0.00\", \"0.88±0.000.88±0.00\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"0.89±0.000.62±0.00\", \"caption\": \"Link prediction performance on seven scRNA-seq datasets with 500 most-variable genes. Each dataset includes a cell-type-specific ground-truth network. The values reported are averages from 50 independent evaluations per cell type. scRegNet utilizing the three backbone models—scBERT, Geneformer, and scFoundation—consistently outperforms the baselines.\"}","artifact_sha256":"65b3272d47bb9c4ee1e7a965169bef63add9dbeb31508d4076e5145b761af4ec","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11838224/fullTextXML"},"legacy_id":"lit-031","legacy_row":{"id":"lit-031","paper_id":"scregnet-2025","domain_id":"cells-tissues","task":"Gene-regulatory link prediction","model":"scRegNet (Geneformer backbone)","model_version":"","dataset":"hESC cell-type-specific GRN","dataset_version":"","split":"","metric":"AUROC","value":"0.89","unit":"unitless","uncertainty":"± 0.00 as printed","protocol":"TFs plus 500 variable genes; mean from 50 independent evaluations.","source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838224/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-032","kind":"result","name":"scRegNet (scBERT backbone) · AUROC · hESC cell-type-specific GRN","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-032"}],"attributes":{"printed_value":"0.88","numeric_value":"0.88","metric":"AUROC","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.00 as printed","source_locator":"Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.542807+00:00","notes":"Read break elements: cells contain AUROC on first line then AUPRC. Selected hESC (first cell type), first line. Caption specifies 500 most-variable genes and 50 independent evaluations. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T2\", \"row_cells\": [\"scRegNet (w/ scBERT)\", \"AUROCAUPRC\", \"0.88±0.000.61±0.00\", \"0.90±0.000.83±0.00\", \"0.75±0.010.12±0.01\", \"0.92±0.000.84±0.00\", \"0.92±0.000.94±0.00\", \"0.92±0.000.93±0.00\", \"0.85±0.010.85±0.01\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"0.88±0.000.61±0.00\", \"caption\": \"Link prediction performance on seven scRNA-seq datasets with 500 most-variable genes. Each dataset includes a cell-type-specific ground-truth network. The values reported are averages from 50 independent evaluations per cell type. scRegNet utilizing the three backbone models—scBERT, Geneformer, and scFoundation—consistently outperforms the baselines.\"}","artifact_sha256":"65b3272d47bb9c4ee1e7a965169bef63add9dbeb31508d4076e5145b761af4ec","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11838224/fullTextXML"},"legacy_id":"lit-032","legacy_row":{"id":"lit-032","paper_id":"scregnet-2025","domain_id":"cells-tissues","task":"Gene-regulatory link prediction","model":"scRegNet (scBERT backbone)","model_version":"","dataset":"hESC cell-type-specific GRN","dataset_version":"","split":"","metric":"AUROC","value":"0.88","unit":"unitless","uncertainty":"± 0.00 as printed","protocol":"TFs plus 500 variable genes; mean from 50 independent evaluations.","source_locator":"Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838224/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-033","kind":"result","name":"ProkBERT-mini · Accuracy · E. coli sigma70 promoter dataset","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-033"}],"attributes":{"printed_value":"0.87","numeric_value":"0.87","metric":"Accuracy","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, ProkBERT-mini row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.197Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, ProkBERT-mini row, Accuracy column; cell: 0.87","artifact_sha256":"8610e2a54aa877c8dc565a9cdb6e82099f284c5e0907a52cab18d994ea732436","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10810988/fullTextXML"},"legacy_id":"lit-033","legacy_row":{"id":"lit-033","paper_id":"prokbert-2024","domain_id":"microbes-communities","task":"E. coli sigma70 promoter prediction","model":"ProkBERT-mini","model_version":"","dataset":"E. coli sigma70 promoter dataset","dataset_version":"","split":"","metric":"Accuracy","value":"0.87","unit":"unitless","uncertainty":"","protocol":"Promoter versus non-promoter classification.","source_locator":"Table 3, ProkBERT-mini row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10810988/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-034","kind":"result","name":"Promotech · Accuracy · E. coli sigma70 promoter dataset","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-034"}],"attributes":{"printed_value":"0.71","numeric_value":"0.71","metric":"Accuracy","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Promotech row, Accuracy column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.197Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, Promotech row, Accuracy column; cell: 0.71","artifact_sha256":"8610e2a54aa877c8dc565a9cdb6e82099f284c5e0907a52cab18d994ea732436","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10810988/fullTextXML"},"legacy_id":"lit-034","legacy_row":{"id":"lit-034","paper_id":"prokbert-2024","domain_id":"microbes-communities","task":"E. coli sigma70 promoter prediction","model":"Promotech","model_version":"","dataset":"E. coli sigma70 promoter dataset","dataset_version":"","split":"","metric":"Accuracy","value":"0.71","unit":"unitless","uncertainty":"","protocol":"Promoter versus non-promoter classification.","source_locator":"Table 3, Promotech row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10810988/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-035","kind":"result","name":"Eco70PromBERT · Promoter-class F1 · Independent E. coli sigma70 test dataset","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-035"}],"attributes":{"printed_value":"0.91","numeric_value":"0.91","metric":"Promoter-class F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.544033+00:00","notes":"F1 score is the third two-column group; selected Promoter subcolumn, not AUROC or precision. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T3\", \"row_cells\": [\"Eco70PromBERT (BERT-base + 1bp tokenizer)\", \"0.92\", \"0.90\", \"0.91\", \"0.91\", \"0.91\", \"0.91\", \"110\", \"108\"], \"selected_cell_zero_based\": 5, \"selected_cell_xml\": \"0.91\", \"caption\": \"Performance of Eco70PromBERT and popular promoter prediction models for E.coli using an independent dataset (σ70 promoters and non-promoters).\"}","artifact_sha256":"74278ccd77b2bc00a3f4434546545e8bdec8b0652a0e5d1862ec0f91decccd8d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9745317/fullTextXML"},"legacy_id":"lit-035","legacy_row":{"id":"lit-035","paper_id":"cyaprombert-2022","domain_id":"microbes-communities","task":"E. coli sigma70 promoter prediction","model":"Eco70PromBERT","model_version":"","dataset":"Independent E. coli sigma70 test dataset","dataset_version":"","split":"independent test","metric":"Promoter-class F1","value":"0.91","unit":"unitless","uncertainty":"","protocol":"BERT-base with 1bp tokenizer; 110 promoters and 108 non-promoters.","source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9745317/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-036","kind":"result","name":"iPro70-FMWin · Promoter-class F1 · Independent E. coli sigma70 test dataset","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-036"}],"attributes":{"printed_value":"0.90","numeric_value":"0.90","metric":"Promoter-class F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"TABLE 3, iPro70-FMWin row, F1 score Promoter column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.544926+00:00","notes":"F1 score is the third two-column group; selected Promoter subcolumn, not AUROC or precision. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"T3\", \"row_cells\": [\"iPro70-FMWin\", \"0.90\", \"0.90\", \"0.93\", \"0.88\", \"0.90\", \"0.91\", \"110\", \"108\"], \"selected_cell_zero_based\": 5, \"selected_cell_xml\": \"0.90\", \"caption\": \"Performance of Eco70PromBERT and popular promoter prediction models for E.coli using an independent dataset (σ70 promoters and non-promoters).\"}","artifact_sha256":"74278ccd77b2bc00a3f4434546545e8bdec8b0652a0e5d1862ec0f91decccd8d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9745317/fullTextXML"},"legacy_id":"lit-036","legacy_row":{"id":"lit-036","paper_id":"cyaprombert-2022","domain_id":"microbes-communities","task":"E. coli sigma70 promoter prediction","model":"iPro70-FMWin","model_version":"","dataset":"Independent E. coli sigma70 test dataset","dataset_version":"","split":"independent test","metric":"Promoter-class F1","value":"0.90","unit":"unitless","uncertainty":"","protocol":"Compared on the same independent test dataset; 110 promoters and 108 non-promoters.","source_locator":"TABLE 3, iPro70-FMWin row, F1 score Promoter column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9745317/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-037","kind":"result","name":"EVO2 · MCC · LAMBDA genome-wide prophage test","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-037"}],"attributes":{"printed_value":"0.680","numeric_value":"0.680","metric":"MCC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 5, EVO2 row, MCC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.240Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, EVO2 row, MCC column; cell: 0.680","artifact_sha256":"22c2e218e87dce757907f6086a0e2ad37c13f785b34fff5bea7cfa1a6c276b16","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13041943/fullTextXML"},"legacy_id":"lit-037","legacy_row":{"id":"lit-037","paper_id":"lambda-prophage-2026","domain_id":"microbes-communities","task":"Genome-wide prophage detection","model":"EVO2","model_version":"","dataset":"LAMBDA genome-wide prophage test","dataset_version":"","split":"","metric":"MCC","value":"0.680","unit":"unitless","uncertainty":"","protocol":"Genomic language model fine-tuned for prophage detection; genome-wide evaluation.","source_locator":"Table 5, EVO2 row, MCC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13041943/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-038","kind":"result","name":"geNomad · MCC · LAMBDA genome-wide prophage test","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-038"}],"attributes":{"printed_value":"0.794","numeric_value":"0.794","metric":"MCC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 5, geNomad row, MCC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:36.240Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, geNomad row, MCC column; cell: 0.794","artifact_sha256":"22c2e218e87dce757907f6086a0e2ad37c13f785b34fff5bea7cfa1a6c276b16","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13041943/fullTextXML"},"legacy_id":"lit-038","legacy_row":{"id":"lit-038","paper_id":"lambda-prophage-2026","domain_id":"microbes-communities","task":"Genome-wide prophage detection","model":"geNomad","model_version":"","dataset":"LAMBDA genome-wide prophage test","dataset_version":"","split":"","metric":"MCC","value":"0.794","unit":"unitless","uncertainty":"","protocol":"Traditional specialist comparator; genome-wide evaluation.","source_locator":"Table 5, geNomad row, MCC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13041943/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-039","kind":"result","name":"NABAS+ · F1 score · CAMI II Toy human gastrooral sample19-new","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-039"}],"attributes":{"printed_value":"0.719","numeric_value":"0.719","metric":"F1 score","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.546107+00:00","notes":"Selected Sample19-new explicitly, not Sample19-old, and F1 rather than precision/recall. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl3\", \"row_cells\": [\"Sample19-new\", \"NABAS+\", \"0.719\", \"0.719\", \"0.719\"], \"selected_cell_zero_based\": 4, \"selected_cell_xml\": \"0.719\", \"caption\": \"Performance of the classifier on the original and newly generated sample19\"}","artifact_sha256":"49903e751beb86f6744825d2fdb3ea2fbe327b52b1ce68c323bfa8b66dae71ec","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12231603/fullTextXML"},"legacy_id":"lit-039","legacy_row":{"id":"lit-039","paper_id":"nabas-plus-2025","domain_id":"microbes-communities","task":"Metagenomic taxonomic classification","model":"NABAS+","model_version":"","dataset":"CAMI II Toy human gastrooral sample19-new","dataset_version":"","split":"","metric":"F1 score","value":"0.719","unit":"unitless","uncertainty":"","protocol":"Newly generated sample19 used for classifier comparison.","source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12231603/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-040","kind":"result","name":"MetaPhlAn3 · F1 score · CAMI II Toy human gastrooral sample19-new","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-040"}],"attributes":{"printed_value":"0.753","numeric_value":"0.753","metric":"F1 score","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Sample19-new / MetaPhlAn3 row, F1 score column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.547023+00:00","notes":"Selected Sample19-new explicitly, not Sample19-old, and F1 rather than precision/recall. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl3\", \"row_cells\": [\"Sample19-new\", \"MetaPhlAn3\", \"0.778\", \"0.729\", \"0.753\"], \"selected_cell_zero_based\": 4, \"selected_cell_xml\": \"0.753\", \"caption\": \"Performance of the classifier on the original and newly generated sample19\"}","artifact_sha256":"49903e751beb86f6744825d2fdb3ea2fbe327b52b1ce68c323bfa8b66dae71ec","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12231603/fullTextXML"},"legacy_id":"lit-040","legacy_row":{"id":"lit-040","paper_id":"nabas-plus-2025","domain_id":"microbes-communities","task":"Metagenomic taxonomic classification","model":"MetaPhlAn3","model_version":"","dataset":"CAMI II Toy human gastrooral sample19-new","dataset_version":"","split":"","metric":"F1 score","value":"0.753","unit":"unitless","uncertainty":"","protocol":"Newly generated sample19 used for classifier comparison.","source_locator":"Table 3, Sample19-new / MetaPhlAn3 row, F1 score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12231603/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-041","kind":"result","name":"Chai-1 · Success rate, ligand all-atom RMSD <2 Å · LiPP lipid–protein complexes","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-041"}],"attributes":{"printed_value":"60.7","numeric_value":"60.7","metric":"Success rate, ligand all-atom RMSD <2 Å","metric_direction":"unknown","unit":"%","uncertainty":"95% CI 55.2–66.0","source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.548973+00:00","notes":"JATS label is bare 2, which caused original parser miss. LiPP N=331 full-set column selected, not N=36 test subset. Caption success is lipid all-atom RMSD <2 Angstrom; PB-valid is a separate table. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl2\", \"row_cells\": [\"Chai-1\", \"60.7 {55.2–66.0}\", \"36.1 {20.8–53.7}\", \"77\", \"-\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"60.7 {55.2–66.0}\", \"caption\": \"Success Rates (Success Defined Only by Lipid Pose All-Atom RMSD Cutoff Values Less Than 2 Å) of the Five Computational Methods Used in This Study on Lipid–Protein Complexes (via LiPP Benchmark Set) Compared to Protein-Small Molecule Complexes (via PoseBusters Benchmark Set)\"}","artifact_sha256":"6ff34f2f709a14858a3753abf9f8f6efa1e7e3c351f15c70cf64264193a9414e","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13292216/fullTextXML"},"legacy_id":"lit-041","legacy_row":{"id":"lit-041","paper_id":"lipp-2026","domain_id":"molecular-interactions","task":"Lipid–protein binding pose","model":"Chai-1","model_version":"","dataset":"LiPP lipid–protein complexes","dataset_version":"331 complexes","split":"","metric":"Success rate, ligand all-atom RMSD <2 Å","value":"60.7","unit":"%","uncertainty":"95% CI 55.2–66.0","protocol":"Top-scoring pose; all-atom lipid RMSD below 2 Å.","source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13292216/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-042","kind":"result","name":"DiffDock-L · Success rate, ligand all-atom RMSD <2 Å · LiPP lipid–protein complexes","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-042"}],"attributes":{"printed_value":"46.8","numeric_value":"46.8","metric":"Success rate, ligand all-atom RMSD <2 Å","metric_direction":"unknown","unit":"%","uncertainty":"95% CI 41.3–52.3","source_locator":"Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.550691+00:00","notes":"JATS label is bare 2, which caused original parser miss. LiPP N=331 full-set column selected, not N=36 test subset. Caption success is lipid all-atom RMSD <2 Angstrom; PB-valid is a separate table. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl2\", \"row_cells\": [\"DiffDock-L\", \"46.8 {41.3–52.3}\", \"30.5 {16.3–48.1}\", \"-\", \"50\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"46.8 {41.3–52.3}\", \"caption\": \"Success Rates (Success Defined Only by Lipid Pose All-Atom RMSD Cutoff Values Less Than 2 Å) of the Five Computational Methods Used in This Study on Lipid–Protein Complexes (via LiPP Benchmark Set) Compared to Protein-Small Molecule Complexes (via PoseBusters Benchmark Set)\"}","artifact_sha256":"6ff34f2f709a14858a3753abf9f8f6efa1e7e3c351f15c70cf64264193a9414e","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13292216/fullTextXML"},"legacy_id":"lit-042","legacy_row":{"id":"lit-042","paper_id":"lipp-2026","domain_id":"molecular-interactions","task":"Lipid–protein binding pose","model":"DiffDock-L","model_version":"","dataset":"LiPP lipid–protein complexes","dataset_version":"331 complexes","split":"","metric":"Success rate, ligand all-atom RMSD <2 Å","value":"46.8","unit":"%","uncertainty":"95% CI 41.3–52.3","protocol":"Top-scoring pose; all-atom lipid RMSD below 2 Å.","source_locator":"Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13292216/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-043","kind":"result","name":"DiffDock-NMDN · Forward-screening success rate · CASF-2016 blind docked poses","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-043"}],"attributes":{"printed_value":"66.7","numeric_value":"66.7","metric":"Forward-screening success rate","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.552697+00:00","notes":"All scoring functions share DiffDock-NMDN poses via rowspan. Selected forward-screening success percentage, not docking pose success or scoring-power correlation. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl2\", \"row_cells\": [\"DiffDock-NMDN\", \"NMDN\", \"35.85\", \"66.7\", \"0.376\", \"0.458\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"66.7\", \"caption\": \"Performance of Representative Scoring Functions on the CASF-2016 Using the DiffDock-NMDN Blind Docked Posesa\"}","artifact_sha256":"194b21478aaedd9a7384cabb8b0040ca5b6a4938f4d275627b86a6b787affc20","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11815853/fullTextXML"},"legacy_id":"lit-043","legacy_row":{"id":"lit-043","paper_id":"nmdn-2025","domain_id":"molecular-interactions","task":"Protein–ligand virtual screening","model":"DiffDock-NMDN","model_version":"","dataset":"CASF-2016 blind docked poses","dataset_version":"","split":"","metric":"Forward-screening success rate","value":"66.7","unit":"%","uncertainty":"","protocol":"NMDN scoring on DiffDock-NMDN blind docked poses; not ligand-pose RMSD.","source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11815853/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-044","kind":"result","name":"Vina · Forward-screening success rate · CASF-2016 blind docked poses","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-044"}],"attributes":{"printed_value":"42.1","numeric_value":"42.1","metric":"Forward-screening success rate","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, Vina scoring row, success rate (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.554518+00:00","notes":"All scoring functions share DiffDock-NMDN poses via rowspan. Selected forward-screening success percentage, not docking pose success or scoring-power correlation. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl2\", \"row_cells\": [\"Vina38\", \"15.18\", \"42.1\", \"0.340\", \"0.323\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"42.1\", \"caption\": \"Performance of Representative Scoring Functions on the CASF-2016 Using the DiffDock-NMDN Blind Docked Posesa\"}","artifact_sha256":"194b21478aaedd9a7384cabb8b0040ca5b6a4938f4d275627b86a6b787affc20","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11815853/fullTextXML"},"legacy_id":"lit-044","legacy_row":{"id":"lit-044","paper_id":"nmdn-2025","domain_id":"molecular-interactions","task":"Protein–ligand virtual screening","model":"Vina","model_version":"","dataset":"CASF-2016 blind docked poses","dataset_version":"","split":"","metric":"Forward-screening success rate","value":"42.1","unit":"%","uncertainty":"","protocol":"Vina scoring on the same DiffDock-NMDN blind docked poses; not ligand-pose RMSD.","source_locator":"Table 2, Vina scoring row, success rate (%) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11815853/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-045","kind":"result","name":"Boltz-1 · Median ligand RMSD · PLINDER-L95","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-045"}],"attributes":{"printed_value":"1.393","numeric_value":"1.393","metric":"Median ligand RMSD","metric_direction":"unknown","unit":"Å","uncertainty":null,"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.555674+00:00","notes":"JATS label is bare 1. Selected Ligand RMSD column in Plinder-L95, not Protein RMSD; retained method row without refinement settings. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl1\", \"row_cells\": [\"Boltz-1\", \"\", \"0.4041\", \"1.393\", \"60.58\", \"0.0283\", \"0.4381\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"1.393\", \"caption\": \"Evaluation Metrics for Protein and Ligand Structure Prediction Performance across All Entries in the Plinder-L95 Data Set\"}","artifact_sha256":"78a77b9a0ab8bfa371f5b9baef3f443f4590d6e71cf864d67e90e9ebdfa7fc1b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12658688/fullTextXML"},"legacy_id":"lit-045","legacy_row":{"id":"lit-045","paper_id":"boltz-stereochemistry-2025","domain_id":"molecular-interactions","task":"Protein–ligand pose prediction","model":"Boltz-1","model_version":"","dataset":"PLINDER-L95","dataset_version":"","split":"","metric":"Median ligand RMSD","value":"1.393","unit":"Å","uncertainty":"","protocol":"All entries; authors note this dataset contains structures seen during model training.","source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12658688/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-046","kind":"result","name":"DiffDock · Median ligand RMSD · PLINDER-L95","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-046"}],"attributes":{"printed_value":"1.342","numeric_value":"1.342","metric":"Median ligand RMSD","metric_direction":"unknown","unit":"Å","uncertainty":null,"source_locator":"Table 1, DiffDock row, Ligand RMSD (Å) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.556572+00:00","notes":"JATS label is bare 1. Selected Ligand RMSD column in Plinder-L95, not Protein RMSD; retained method row without refinement settings. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl1\", \"row_cells\": [\"DiffDock\", \"\", \"\", \"1.342\", \"100\", \"0\", \"0\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"1.342\", \"caption\": \"Evaluation Metrics for Protein and Ligand Structure Prediction Performance across All Entries in the Plinder-L95 Data Set\"}","artifact_sha256":"78a77b9a0ab8bfa371f5b9baef3f443f4590d6e71cf864d67e90e9ebdfa7fc1b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12658688/fullTextXML"},"legacy_id":"lit-046","legacy_row":{"id":"lit-046","paper_id":"boltz-stereochemistry-2025","domain_id":"molecular-interactions","task":"Protein–ligand pose prediction","model":"DiffDock","model_version":"","dataset":"PLINDER-L95","dataset_version":"","split":"","metric":"Median ligand RMSD","value":"1.342","unit":"Å","uncertainty":"","protocol":"All entries; rigid-protein docking comparator; authors note this dataset contains structures seen during model training.","source_locator":"Table 1, DiffDock row, Ligand RMSD (Å) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12658688/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-047","kind":"result","name":"Boltz-2 · Pearson R · SARS-CoV-2 Mpro ligands","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-047"}],"attributes":{"printed_value":"0.800","numeric_value":"0.800","metric":"Pearson R","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.027","source_locator":"Table 3, Boltz-2 row, Pearson’s R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.557756+00:00","notes":"JATS label is bare 3. Selected SARS-CoV-2 Mpro potency Pearson R, not MERS-CoV table 2 or Boltz-2-Internal row; uncertainty retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl3\", \"row_cells\": [\"Boltz-2\", \"0.716 ± 0.036\", \"0.909 ± 0.043\", \"0.800 ± 0.027\", \"9.54 × 10–59\", \"0.532 ± 0.043\", \"0.598 ± 0.024\", \"0.799 ± 0.012\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"0.800 ± 0.027\", \"caption\": \"Statistical Performance Metrics of Potency Prediction for SARS-CoV-2 Mpro Using Different Ligand Pose Generation Protocols\"}","artifact_sha256":"c356a1c65a0033e5ae18a05d4afab5495856c5b6869328ff49e13547a4801a57","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12801289/fullTextXML"},"legacy_id":"lit-047","legacy_row":{"id":"lit-047","paper_id":"mpro-pose-affinity-2025","domain_id":"molecular-interactions","task":"Ligand potency prediction using generated poses","model":"Boltz-2","model_version":"","dataset":"SARS-CoV-2 Mpro ligands","dataset_version":"","split":"","metric":"Pearson R","value":"0.800","unit":"unitless","uncertainty":"± 0.027","protocol":"Potency prediction using Boltz-2 ligand-pose generation protocol; see paper scoring pipeline.","source_locator":"Table 3, Boltz-2 row, Pearson’s R column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12801289/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-048","kind":"result","name":"DiffDock · Pearson R · SARS-CoV-2 Mpro ligands","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-048"}],"attributes":{"printed_value":"0.695","numeric_value":"0.695","metric":"Pearson R","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.037","source_locator":"Table 3, DiffDock row, Pearson’s R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:41:16.558815+00:00","notes":"JATS label is bare 3. Selected SARS-CoV-2 Mpro potency Pearson R, not MERS-CoV table 2 or Boltz-2-Internal row; uncertainty retained in evidence. This verifies the central score at its source location, not every metadata field or an experimental reproduction.","evidence":"{\"table_xml_id\": \"tbl3\", \"row_cells\": [\"DiffDock\", \"0.973 ± 0.044\", \"1.192 ± 0.045\", \"0.695 ± 0.037\", \"9.93 × 10–39\", \"0.195 ± 0.061\", \"0.512 ± 0.028\", \"0.756 ± 0.014\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"0.695 ± 0.037\", \"caption\": \"Statistical Performance Metrics of Potency Prediction for SARS-CoV-2 Mpro Using Different Ligand Pose Generation Protocols\"}","artifact_sha256":"c356a1c65a0033e5ae18a05d4afab5495856c5b6869328ff49e13547a4801a57","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12801289/fullTextXML"},"legacy_id":"lit-048","legacy_row":{"id":"lit-048","paper_id":"mpro-pose-affinity-2025","domain_id":"molecular-interactions","task":"Ligand potency prediction using generated poses","model":"DiffDock","model_version":"","dataset":"SARS-CoV-2 Mpro ligands","dataset_version":"","split":"","metric":"Pearson R","value":"0.695","unit":"unitless","uncertainty":"± 0.037","protocol":"Potency prediction using DiffDock ligand-pose generation plus paper scoring pipeline; not a native DiffDock affinity score.","source_locator":"Table 3, DiffDock row, Pearson’s R column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12801289/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:25:00Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-003","kind":"result","name":"Mouse-Geneformer · F1 · Human thymus scRNA-seq","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-003"}],"attributes":{"printed_value":"48.57","numeric_value":"48.57","metric":"F1","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.392488+00:00","notes":"h/ Thymus row, four human cell types; zero-shot model blocks use Acc then F1, not fine-tuned scores. Ortholog-based conversion evaluated, not mouse cell annotation. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"pgen.1011420.t004\", \"row_cells\": [\"h/ Thymus\", \"4\", \"82.74\", \"48.57\", \"95.59\", \"87.81\", \"87.27\", \"74.48\", \"95.44\", \"87.79\", \"92.04\", \"82.37\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"48.57\", \"caption\": \"Human cell type classification using mouse-Geneformer via ortholog-based gene name conversion, compared to native human models (human-Geneformer and scGPT).\"}","artifact_sha256":"ef6c68f5c9b47c2f89598ddf647f05b72e4155e8b33609ebff23936a84bbd41d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11964219/fullTextXML"},"legacy_id":"lit-b3-003","legacy_row":{"id":"lit-b3-003","paper_id":"mouse-geneformer-2025","domain_id":"cells-tissues","task":"Human thymus cell-type classification","model":"Mouse-Geneformer","model_version":"","dataset":"Human thymus scRNA-seq","dataset_version":"","split":"","metric":"F1","value":"48.57","unit":"%","uncertainty":"","protocol":"Ortholog-based gene conversion; zero-shot mouse model on human cells.","source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11964219/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-004","kind":"result","name":"Human-Geneformer · F1 · Human thymus scRNA-seq","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-004"}],"attributes":{"printed_value":"74.48","numeric_value":"74.48","metric":"F1","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.394073+00:00","notes":"h/ Thymus row, four human cell types; zero-shot model blocks use Acc then F1, not fine-tuned scores. Ortholog-based conversion evaluated, not mouse cell annotation. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"pgen.1011420.t004\", \"row_cells\": [\"h/ Thymus\", \"4\", \"82.74\", \"48.57\", \"95.59\", \"87.81\", \"87.27\", \"74.48\", \"95.44\", \"87.79\", \"92.04\", \"82.37\"], \"selected_cell_zero_based\": 7, \"selected_cell_xml\": \"74.48\", \"caption\": \"Human cell type classification using mouse-Geneformer via ortholog-based gene name conversion, compared to native human models (human-Geneformer and scGPT).\"}","artifact_sha256":"ef6c68f5c9b47c2f89598ddf647f05b72e4155e8b33609ebff23936a84bbd41d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11964219/fullTextXML"},"legacy_id":"lit-b3-004","legacy_row":{"id":"lit-b3-004","paper_id":"mouse-geneformer-2025","domain_id":"cells-tissues","task":"Human thymus cell-type classification","model":"Human-Geneformer","model_version":"","dataset":"Human thymus scRNA-seq","dataset_version":"","split":"","metric":"F1","value":"74.48","unit":"%","uncertainty":"","protocol":"Native human model; zero-shot setting.","source_locator":"Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11964219/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-005","kind":"result","name":"scLLMDA · F1 · MosA1 reference → WholeBrainA query","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-005"}],"attributes":{"printed_value":"0.6525","numeric_value":"0.6525","metric":"F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.395850+00:00","notes":"First reference/query block MosA1 to WholeBrainA, F1 second column in block; direction of transfer is part of protocol. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"pcbi.1014226.t002\", \"row_cells\": [\"scLLMDA\", \"0.7228\", \"0.6525\", \"0.7691\", \"0.3993\", \"0.7114\", \"0.6253\", \"0.7044\", \"0.3630\", \"0.7012\", \"0.6490\", \"0.7583\", \"0.4580\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"\\n0.6525\\n\", \"caption\": \"Cell type annotation between snATAC-seq and sciATAC-seq platforms.\"}","artifact_sha256":"f1cdc7d54c6b2d491e4a74a44a1188a3679c555262c988e95a1c0de4614fe3cb","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13132462/fullTextXML"},"legacy_id":"lit-b3-005","legacy_row":{"id":"lit-b3-005","paper_id":"scatac-llmda-2026","domain_id":"cells-tissues","task":"Cross-platform scATAC cell-type annotation","model":"scLLMDA","model_version":"","dataset":"MosA1 reference → WholeBrainA query","dataset_version":"","split":"","metric":"F1","value":"0.6525","unit":"unitless","uncertainty":"","protocol":"Cross-platform reference-query cell-type annotation.","source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13132462/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-006","kind":"result","name":"MINGLE · F1 · MosA1 reference → WholeBrainA query","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-006"}],"attributes":{"printed_value":"0.6256","numeric_value":"0.6256","metric":"F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.397272+00:00","notes":"First reference/query block MosA1 to WholeBrainA, F1 second column in block; direction of transfer is part of protocol. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"pcbi.1014226.t002\", \"row_cells\": [\"MINGLE\", \"0.7143\", \"0.6256\", \"0.7095\", \"0.32\", \"0.7097\", \"0.6243\", \"0.6459\", \"0.2973\", \"0.6665\", \"0.6022\", \"0.7033\", \"0.3756\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"0.6256\", \"caption\": \"Cell type annotation between snATAC-seq and sciATAC-seq platforms.\"}","artifact_sha256":"f1cdc7d54c6b2d491e4a74a44a1188a3679c555262c988e95a1c0de4614fe3cb","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13132462/fullTextXML"},"legacy_id":"lit-b3-006","legacy_row":{"id":"lit-b3-006","paper_id":"scatac-llmda-2026","domain_id":"cells-tissues","task":"Cross-platform scATAC cell-type annotation","model":"MINGLE","model_version":"","dataset":"MosA1 reference → WholeBrainA query","dataset_version":"","split":"","metric":"F1","value":"0.6256","unit":"unitless","uncertainty":"","protocol":"Cross-platform reference-query comparator.","source_locator":"Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13132462/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-011","kind":"result","name":"GenePT-w · Adjusted Rand Index · Aorta single-cell dataset","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-011"}],"attributes":{"printed_value":"0.54","numeric_value":"0.54","metric":"Adjusted Rand Index","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.399274+00:00","notes":"First Cell type row belongs to Aorta, not preceding Phenotype row or later organs. ARI is first in each three-metric method block, not AMI/ASW. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"T6\", \"row_cells\": [\"\", \"Cell type\", \"0.21\", \"0.31\", \"−0.04\", \"0.47\", \"0.64\", \"0.18\", \"0.54\", \"0.60\", \"0.03\", \"0.31\", \"0.47\", \"0.04\"], \"selected_cell_zero_based\": 8, \"selected_cell_xml\": \"\\n0.54\\n\", \"caption\": \"Assessing the Association Between Different Latent Cell Representations and Biological Annotations.This analysis involves datasets representing cells from circulatory systems (Aorta and Artery), bone tissues (Bones, Myeloid), the Pancreas, and immune cells collected from healthy individuals and patients with Multiple Sclerosis. We utilized pretrained Geneformer and scGPT embeddings for this task. The Adjusted Rand Index (ARI) and Adjusted Mutual Information (AMI) were computed to compare the labels derived from k-means clustering with the true annotations of the original samples (higher values indicate better alignment); the Average Silhouette Width (ASW) was calculated using the true annotations of original samples to assess the cohesion and separation of the clusters.\"}","artifact_sha256":"230a2ec55458d9243eaeeebf3244df7409eb02d47f4b809ee56a06dcb6fdd047","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10614824/fullTextXML"},"legacy_id":"lit-b3-011","legacy_row":{"id":"lit-b3-011","paper_id":"genept-2024","domain_id":"cells-tissues","task":"Cell-type structure in frozen embeddings","model":"GenePT-w","model_version":"","dataset":"Aorta single-cell dataset","dataset_version":"","split":"","metric":"Adjusted Rand Index","value":"0.54","unit":"unitless","uncertainty":"","protocol":"k-means on pretrained cell embeddings; agreement with original cell-type labels.","source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10614824/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-012","kind":"result","name":"scGPT · Adjusted Rand Index · Aorta single-cell dataset","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-012"}],"attributes":{"printed_value":"0.47","numeric_value":"0.47","metric":"Adjusted Rand Index","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, Aorta / Cell type row, scGPT ARI column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.402215+00:00","notes":"First Cell type row belongs to Aorta, not preceding Phenotype row or later organs. ARI is first in each three-metric method block, not AMI/ASW. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"T6\", \"row_cells\": [\"\", \"Cell type\", \"0.21\", \"0.31\", \"−0.04\", \"0.47\", \"0.64\", \"0.18\", \"0.54\", \"0.60\", \"0.03\", \"0.31\", \"0.47\", \"0.04\"], \"selected_cell_zero_based\": 5, \"selected_cell_xml\": \"0.47\", \"caption\": \"Assessing the Association Between Different Latent Cell Representations and Biological Annotations.This analysis involves datasets representing cells from circulatory systems (Aorta and Artery), bone tissues (Bones, Myeloid), the Pancreas, and immune cells collected from healthy individuals and patients with Multiple Sclerosis. We utilized pretrained Geneformer and scGPT embeddings for this task. The Adjusted Rand Index (ARI) and Adjusted Mutual Information (AMI) were computed to compare the labels derived from k-means clustering with the true annotations of the original samples (higher values indicate better alignment); the Average Silhouette Width (ASW) was calculated using the true annotations of original samples to assess the cohesion and separation of the clusters.\"}","artifact_sha256":"230a2ec55458d9243eaeeebf3244df7409eb02d47f4b809ee56a06dcb6fdd047","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10614824/fullTextXML"},"legacy_id":"lit-b3-012","legacy_row":{"id":"lit-b3-012","paper_id":"genept-2024","domain_id":"cells-tissues","task":"Cell-type structure in frozen embeddings","model":"scGPT","model_version":"","dataset":"Aorta single-cell dataset","dataset_version":"","split":"","metric":"Adjusted Rand Index","value":"0.47","unit":"unitless","uncertainty":"","protocol":"k-means on pretrained cell embeddings; agreement with original cell-type labels.","source_locator":"Table 2, Aorta / Cell type row, scGPT ARI column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10614824/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-013","kind":"result","name":"Best frozen single-cell foundation model · Balanced accuracy · AIDA v2 PBMC cohort","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-013"}],"attributes":{"printed_value":"0.322","numeric_value":"0.322","metric":"Balanced accuracy","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.008 standard deviation","source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:44.421Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, AIDA v2 row, scFM BA ± SD column; cell: 0.322 ± 0.008","artifact_sha256":"d6cfb13933ceefed630954f804e7dc979b747bc4aec4c8c5518232f25772736a","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13407579/fullTextXML"},"legacy_id":"lit-b3-013","legacy_row":{"id":"lit-b3-013","paper_id":"single-cell-aging-probes-2026","domain_id":"cells-tissues","task":"Donor-aware age-class prediction","model":"Best frozen single-cell foundation model","model_version":"","dataset":"AIDA v2 PBMC cohort","dataset_version":"622 donors","split":"","metric":"Balanced accuracy","value":"0.322","unit":"unitless","uncertainty":"± 0.008 standard deviation","protocol":"Same donor-aware splits and logistic-regression probe as expression PCA; text names Geneformer as best model on AIDA v2.","source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13407579/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-014","kind":"result","name":"Gene-expression PCA · Balanced accuracy · AIDA v2 PBMC cohort","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-014"}],"attributes":{"printed_value":"0.384","numeric_value":"0.384","metric":"Balanced accuracy","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, AIDA v2 row, Gene-expr BA column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:44.421Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, AIDA v2 row, Gene-expr BA column; cell: 0.384","artifact_sha256":"d6cfb13933ceefed630954f804e7dc979b747bc4aec4c8c5518232f25772736a","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13407579/fullTextXML"},"legacy_id":"lit-b3-014","legacy_row":{"id":"lit-b3-014","paper_id":"single-cell-aging-probes-2026","domain_id":"cells-tissues","task":"Donor-aware age-class prediction","model":"Gene-expression PCA","model_version":"","dataset":"AIDA v2 PBMC cohort","dataset_version":"622 donors","split":"","metric":"Balanced accuracy","value":"0.384","unit":"unitless","uncertainty":"","protocol":"Fifty-component gene-expression PCA with the same donor-aware probe splits.","source_locator":"Table 2, AIDA v2 row, Gene-expr BA column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13407579/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-015","kind":"result","name":"scaLR · Cell-type accuracy · PBMCs-BS","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-015"}],"attributes":{"printed_value":"0.942","numeric_value":"0.942","metric":"Cell-type accuracy","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, scaLR row, Cell type Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.403844+00:00","notes":"PBMCs-BS all-feature/all-sample cell-type accuracy block, not cell-state accuracy or time. Footnote letters on model labels excluded from identity. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"TB2\", \"row_cells\": [\"scaLRa\", \"0.942\", \"27:50\", \"9.778\", \"0.840\", \"40:50\", \"8.679\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"\\n0.942\\n\", \"caption\": \"Accuracy, wall-clock time, and memory usage of different pipelines executed using all features and samples from the PBMCs-BS dataset.\"}","artifact_sha256":"829afab6a4e30997c608745d3eb280105b8c5c4e601020ffdc55c866144527ca","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12121358/fullTextXML"},"legacy_id":"lit-b3-015","legacy_row":{"id":"lit-b3-015","paper_id":"scalr-2025","domain_id":"cells-tissues","task":"PBMC cell-type classification","model":"scaLR","model_version":"","dataset":"PBMCs-BS","dataset_version":"","split":"","metric":"Cell-type accuracy","value":"0.942","unit":"unitless","uncertainty":"","protocol":"All features and samples from PBMCs-BS.","source_locator":"Table 2, scaLR row, Cell type Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12121358/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-016","kind":"result","name":"scVI + scANVI · Cell-type accuracy · PBMCs-BS","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-016"}],"attributes":{"printed_value":"0.939","numeric_value":"0.939","metric":"Cell-type accuracy","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.405089+00:00","notes":"PBMCs-BS all-feature/all-sample cell-type accuracy block, not cell-state accuracy or time. Footnote letters on model labels excluded from identity. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"TB2\", \"row_cells\": [\"Svi-tools(scVI & scANVI)b\", \"0.939\", \"53:52\", \"23.914\", \"0.870\", \"54:24\", \"23.645\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"0.939\", \"caption\": \"Accuracy, wall-clock time, and memory usage of different pipelines executed using all features and samples from the PBMCs-BS dataset.\"}","artifact_sha256":"829afab6a4e30997c608745d3eb280105b8c5c4e601020ffdc55c866144527ca","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12121358/fullTextXML"},"legacy_id":"lit-b3-016","legacy_row":{"id":"lit-b3-016","paper_id":"scalr-2025","domain_id":"cells-tissues","task":"PBMC cell-type classification","model":"scVI + scANVI","model_version":"","dataset":"PBMCs-BS","dataset_version":"","split":"","metric":"Cell-type accuracy","value":"0.939","unit":"unitless","uncertainty":"","protocol":"All features and samples from PBMCs-BS; comparison pipeline combines scVI and scANVI.","source_locator":"Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12121358/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-017","kind":"result","name":"scXDR · AUC · scXDR transfer scenario 2","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-017"}],"attributes":{"printed_value":"0.8248","numeric_value":"0.8248","metric":"AUC","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.1573 standard deviation","source_locator":"Table 2, scXDR row, Scenario 2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.056Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, scXDR row, Scenario 2 column; cell: 0.8248 0.1573 ±","artifact_sha256":"47b5925e9887d87fc8288d29288802b1d67d54f064d913151df92171f7c68d33","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12859067/fullTextXML"},"legacy_id":"lit-b3-017","legacy_row":{"id":"lit-b3-017","paper_id":"scxdr-2026","domain_id":"cells-tissues","task":"Cross-dataset single-cell drug response transfer","model":"scXDR","model_version":"","dataset":"scXDR transfer scenario 2","dataset_version":"","split":"","metric":"AUC","value":"0.8248","unit":"unitless","uncertainty":"± 0.1573 standard deviation","protocol":"Single-cell-to-single-cell transfer; source scenario 2.","source_locator":"Table 2, scXDR row, Scenario 2 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12859067/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-018","kind":"result","name":"scVI · AUC · scXDR transfer scenario 2","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-018"}],"attributes":{"printed_value":"0.6970","numeric_value":"0.6970","metric":"AUC","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.2463 standard deviation","source_locator":"Table 2, scVI row, Scenario 2 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.056Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, scVI row, Scenario 2 column; cell: 0.6970 ± 0.2463","artifact_sha256":"47b5925e9887d87fc8288d29288802b1d67d54f064d913151df92171f7c68d33","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12859067/fullTextXML"},"legacy_id":"lit-b3-018","legacy_row":{"id":"lit-b3-018","paper_id":"scxdr-2026","domain_id":"cells-tissues","task":"Cross-dataset single-cell drug response transfer","model":"scVI","model_version":"","dataset":"scXDR transfer scenario 2","dataset_version":"","split":"","metric":"AUC","value":"0.6970","unit":"unitless","uncertainty":"± 0.2463 standard deviation","protocol":"Single-cell-to-single-cell transfer; source scenario 2.","source_locator":"Table 2, scVI row, Scenario 2 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12859067/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-019","kind":"result","name":"CAMMiQ · L1 abundance error · HumanGut-all strain-level query","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-019"}],"attributes":{"printed_value":"0.0517","numeric_value":"0.0517","metric":"L1 abundance error","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.408237+00:00","notes":"HumanGut-all in B. L1 Err. block (second occurrence), not strain count or C. L2 Err. Numbers are errors; lower is better. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"Tab5\", \"row_cells\": [\"\", \"HumanGut-all\", \"0.0517\", \"0.2841\", \"0.2426\", \"0.4004\", \"0.2811\", \"0.4439\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"0.0517\", \"caption\": \"CAMMiQ’s strain level performance compared to Kraken2, KrakenUniq, CLARK, Centrifuge, and MetaPhlAn2, on the four strain-level queries\"}","artifact_sha256":"f0939647ed3de995d58254f79472a612c21b0e1b2560a82783302aa1a148dde3","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9616933/fullTextXML"},"legacy_id":"lit-b3-019","legacy_row":{"id":"lit-b3-019","paper_id":"cammiq-2022","domain_id":"microbes-communities","task":"Strain-level abundance quantification","model":"CAMMiQ","model_version":"","dataset":"HumanGut-all strain-level query","dataset_version":"","split":"","metric":"L1 abundance error","value":"0.0517","unit":"unitless","uncertainty":"","protocol":"Strain-level quantification on the HumanGut-all synthetic query.","source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9616933/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-020","kind":"result","name":"Kraken2 · L1 abundance error · HumanGut-all strain-level query","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-020"}],"attributes":{"printed_value":"0.2841","numeric_value":"0.2841","metric":"L1 abundance error","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.411180+00:00","notes":"HumanGut-all in B. L1 Err. block (second occurrence), not strain count or C. L2 Err. Numbers are errors; lower is better. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"Tab5\", \"row_cells\": [\"\", \"HumanGut-all\", \"0.0517\", \"0.2841\", \"0.2426\", \"0.4004\", \"0.2811\", \"0.4439\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"0.2841\", \"caption\": \"CAMMiQ’s strain level performance compared to Kraken2, KrakenUniq, CLARK, Centrifuge, and MetaPhlAn2, on the four strain-level queries\"}","artifact_sha256":"f0939647ed3de995d58254f79472a612c21b0e1b2560a82783302aa1a148dde3","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9616933/fullTextXML"},"legacy_id":"lit-b3-020","legacy_row":{"id":"lit-b3-020","paper_id":"cammiq-2022","domain_id":"microbes-communities","task":"Strain-level abundance quantification","model":"Kraken2","model_version":"","dataset":"HumanGut-all strain-level query","dataset_version":"","split":"","metric":"L1 abundance error","value":"0.2841","unit":"unitless","uncertainty":"","protocol":"Strain-level quantification on the HumanGut-all synthetic query.","source_locator":"Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9616933/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-021","kind":"result","name":"Lazypipe-nt · Genus-level F1 · Simulated viral metagenome","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-021"}],"attributes":{"printed_value":"0.932","numeric_value":"0.932","metric":"Genus-level F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, Lazypipe-nt / Genus row, F column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.412605+00:00","notes":"First Genus block selected using rank row span, not Species. Final F column is F score, not precision or recall. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"veaa091-T1\", \"row_cells\": [\"Lazypipe-nt\", \"Genus\", \"41\", \"2\", \"4\", \"0.953\", \"0.911\", \"0.932\"], \"selected_cell_zero_based\": -1, \"selected_cell_xml\": \"0.932\", \"caption\": \"Accessing accuracy of virus taxon retrieval by different tools.\"}","artifact_sha256":"77842d8e4f6b419e331ab5a01fdf8f9eb8604f259425d79602be896aad3d0ad1","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7772471/fullTextXML"},"legacy_id":"lit-b3-021","legacy_row":{"id":"lit-b3-021","paper_id":"lazypipe-2020","domain_id":"microbes-communities","task":"Simulated metagenome virus-taxon retrieval","model":"Lazypipe-nt","model_version":"","dataset":"Simulated viral metagenome","dataset_version":"","split":"","metric":"Genus-level F1","value":"0.932","unit":"unitless","uncertainty":"","protocol":"Genus-rank viral taxon retrieval.","source_locator":"Table 1, Lazypipe-nt / Genus row, F column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7772471/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-022","kind":"result","name":"Kraken2 · Genus-level F1 · Simulated viral metagenome","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-022"}],"attributes":{"printed_value":"0.627","numeric_value":"0.627","metric":"Genus-level F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, Kraken2 / Genus row, F column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.413635+00:00","notes":"First Genus block selected using rank row span, not Species. Final F column is F score, not precision or recall. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"veaa091-T1\", \"row_cells\": [\"Kraken2\", \"21\", \"1\", \"24\", \"0.955\", \"0.467\", \"0.627\"], \"selected_cell_zero_based\": -1, \"selected_cell_xml\": \"0.627\", \"caption\": \"Accessing accuracy of virus taxon retrieval by different tools.\"}","artifact_sha256":"77842d8e4f6b419e331ab5a01fdf8f9eb8604f259425d79602be896aad3d0ad1","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7772471/fullTextXML"},"legacy_id":"lit-b3-022","legacy_row":{"id":"lit-b3-022","paper_id":"lazypipe-2020","domain_id":"microbes-communities","task":"Simulated metagenome virus-taxon retrieval","model":"Kraken2","model_version":"","dataset":"Simulated viral metagenome","dataset_version":"","split":"","metric":"Genus-level F1","value":"0.627","unit":"unitless","uncertainty":"","protocol":"Genus-rank viral taxon retrieval.","source_locator":"Table 1, Kraken2 / Genus row, F column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7772471/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-023","kind":"result","name":"NCD-gzip · Macro F1 · CAMI II Sample_0 10,000-read subsample","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["CAMI II superkingdom read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-023"}],"attributes":{"printed_value":"0.9804","numeric_value":"0.9804","metric":"Macro F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 5, NCD Superkingdom row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.414806+00:00","notes":"NCD rank-specific table 5, F1 column; Superkingdom and Phylum are different classification granularities. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"table-5\", \"row_cells\": [\"Superkingdom\", \"0.9616\", \"1.0000\", \"0.9804\", \"0.9616\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"0.9804\", \"caption\": \"Taxonomic classification on the CAMI II 10,000-read subsample (Sample_0).Metrics are macro-averaged (recall, precision, F1) and micro-averaged (accuracy). NCD uses genome fragmentation (‘Genome fragmentation’) and assigns every read; Kraken2 uses low-confidence assignments and leaves 61.4% unclassified.\"}","artifact_sha256":"10e9ba780c45e7787baff9b81ef7b45c014d7fe6c716d6c759e14d89a813dc1c","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12884959/fullTextXML"},"legacy_id":"lit-b3-023","legacy_row":{"id":"lit-b3-023","paper_id":"ncd-metagenomics-2026","domain_id":"microbes-communities","task":"CAMI II superkingdom read classification","model":"NCD-gzip","model_version":"","dataset":"CAMI II Sample_0 10,000-read subsample","dataset_version":"10,000 reads","split":"","metric":"Macro F1","value":"0.9804","unit":"unitless","uncertainty":"","protocol":"Superkingdom-level macro-averaged F1; NCD assigns every read.","source_locator":"Table 5, NCD Superkingdom row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12884959/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-024","kind":"result","name":"NCD-gzip · Macro F1 · CAMI II Sample_0 10,000-read subsample","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["CAMI II phylum read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-024"}],"attributes":{"printed_value":"0.1263","numeric_value":"0.1263","metric":"Macro F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 5, NCD Phylum row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.415788+00:00","notes":"NCD rank-specific table 5, F1 column; Superkingdom and Phylum are different classification granularities. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"table-5\", \"row_cells\": [\"Phylum\", \"0.1336\", \"0.1637\", \"0.1263\", \"0.2709\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"0.1263\", \"caption\": \"Taxonomic classification on the CAMI II 10,000-read subsample (Sample_0).Metrics are macro-averaged (recall, precision, F1) and micro-averaged (accuracy). NCD uses genome fragmentation (‘Genome fragmentation’) and assigns every read; Kraken2 uses low-confidence assignments and leaves 61.4% unclassified.\"}","artifact_sha256":"10e9ba780c45e7787baff9b81ef7b45c014d7fe6c716d6c759e14d89a813dc1c","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12884959/fullTextXML"},"legacy_id":"lit-b3-024","legacy_row":{"id":"lit-b3-024","paper_id":"ncd-metagenomics-2026","domain_id":"microbes-communities","task":"CAMI II phylum read classification","model":"NCD-gzip","model_version":"","dataset":"CAMI II Sample_0 10,000-read subsample","dataset_version":"10,000 reads","split":"","metric":"Macro F1","value":"0.1263","unit":"unitless","uncertainty":"","protocol":"Phylum-level macro-averaged F1; distinct taxonomic rank from the other row.","source_locator":"Table 5, NCD Phylum row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12884959/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-025","kind":"result","name":"VIBRANT · Average prophage F1 · 20 medium/high-complexity viral simulations","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-025"}],"attributes":{"printed_value":"0.169","numeric_value":"0.169","metric":"Average prophage F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, Vibrant row, Prophage F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.134Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, Vibrant row, Prophage F1 column; cell: 0.169","artifact_sha256":"93a24652edfa6d9f686862479df3addf50a2d5d432d2d5a31882075fa59dfdfd","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8207588/fullTextXML"},"legacy_id":"lit-b3-025","legacy_row":{"id":"lit-b3-025","paper_id":"viral-contig-simulation-2021","domain_id":"microbes-communities","task":"Simulated prophage-contig detection","model":"VIBRANT","model_version":"","dataset":"20 medium/high-complexity viral simulations","dataset_version":"","split":"","metric":"Average prophage F1","value":"0.169","unit":"unitless","uncertainty":"","protocol":"Average across twenty medium- and high-complexity simulated communities.","source_locator":"Table 3, Vibrant row, Prophage F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8207588/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-026","kind":"result","name":"VirSorter · Average prophage F1 · 20 medium/high-complexity viral simulations","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-026"}],"attributes":{"printed_value":"0.147","numeric_value":"0.147","metric":"Average prophage F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, VirSorter row, Prophage F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:50.134Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, VirSorter row, Prophage F1 column; cell: 0.147","artifact_sha256":"93a24652edfa6d9f686862479df3addf50a2d5d432d2d5a31882075fa59dfdfd","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8207588/fullTextXML"},"legacy_id":"lit-b3-026","legacy_row":{"id":"lit-b3-026","paper_id":"viral-contig-simulation-2021","domain_id":"microbes-communities","task":"Simulated prophage-contig detection","model":"VirSorter","model_version":"","dataset":"20 medium/high-complexity viral simulations","dataset_version":"","split":"","metric":"Average prophage F1","value":"0.147","unit":"unitless","uncertainty":"","protocol":"Average across twenty medium- and high-complexity simulated communities.","source_locator":"Table 3, VirSorter row, Prophage F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8207588/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-027","kind":"result","name":"GenomeOcean · F1 · GenomeOcean natural/artificial sequence test","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-027"}],"attributes":{"printed_value":"99.03","numeric_value":"99.03","metric":"F1","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, GenomeOcean row, F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.224Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, GenomeOcean row, F1 column; cell: 99.03","artifact_sha256":"3cc0df52522fccda23e3958f069c916b87ee50bb5c9a992fa37e25256546e145","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11838515/fullTextXML"},"legacy_id":"lit-b3-027","legacy_row":{"id":"lit-b3-027","paper_id":"genomeocean-2025","domain_id":"microbes-communities","task":"Natural vs artificial microbial genome sequence","model":"GenomeOcean","model_version":"","dataset":"GenomeOcean natural/artificial sequence test","dataset_version":"","split":"","metric":"F1","value":"99.03","unit":"%","uncertainty":"","protocol":"Source reports natural-versus-artificial sequence classification.","source_locator":"Table 2, GenomeOcean row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838515/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-028","kind":"result","name":"DNABERT-2 · F1 · GenomeOcean natural/artificial sequence test","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-028"}],"attributes":{"printed_value":"85.12","numeric_value":"85.12","metric":"F1","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, DNABERT-2 row, F1 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.224Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, DNABERT-2 row, F1 column; cell: 85.12","artifact_sha256":"3cc0df52522fccda23e3958f069c916b87ee50bb5c9a992fa37e25256546e145","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11838515/fullTextXML"},"legacy_id":"lit-b3-028","legacy_row":{"id":"lit-b3-028","paper_id":"genomeocean-2025","domain_id":"microbes-communities","task":"Natural vs artificial microbial genome sequence","model":"DNABERT-2","model_version":"","dataset":"GenomeOcean natural/artificial sequence test","dataset_version":"","split":"","metric":"F1","value":"85.12","unit":"%","uncertainty":"","protocol":"Source reports natural-versus-artificial sequence classification.","source_locator":"Table 2, DNABERT-2 row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838515/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-029","kind":"result","name":"kMetaShot · Genus-level F1 · Real mock community MAGs","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-029"}],"attributes":{"printed_value":"95.83","numeric_value":"95.83","metric":"Genus-level F1","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, F1-score % row, Genus kMS column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.417367+00:00","notes":"Real mock sequencing table, F1-score percentage row; Genus is final three-column block, selecting kMS or Gtk rather than Species/Strain. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"TB2\", \"row_cells\": [\"F1-score %\", \"28.13\", \"85.42\", \"86.24\", \"87.18\", \"87.72\", \"95.83\", \"89.80\", \"94.85\"], \"selected_cell_zero_based\": 6, \"selected_cell_xml\": \"95.83\", \"caption\": \"Classification performance metrics measured on the real mock sequencing data.\"}","artifact_sha256":"4584e93ea035c1170b8756a0a52cbe99fe72e70bd09b5f1dee639ee104f78247","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11695915/fullTextXML"},"legacy_id":"lit-b3-029","legacy_row":{"id":"lit-b3-029","paper_id":"kmetashot-2025","domain_id":"microbes-communities","task":"Mock-community MAG taxonomy classification","model":"kMetaShot","model_version":"","dataset":"Real mock community MAGs","dataset_version":"","split":"","metric":"Genus-level F1","value":"95.83","unit":"%","uncertainty":"","protocol":"Genus classification of MAGs from MegaHIT contigs; uncorrected kMetaShot.","source_locator":"Table 2, F1-score % row, Genus kMS column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11695915/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-030","kind":"result","name":"GTDB-Tk · Genus-level F1 · Real mock community MAGs","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-030"}],"attributes":{"printed_value":"89.80","numeric_value":"89.80","metric":"Genus-level F1","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, F1-score % row, Genus Gtk column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.418914+00:00","notes":"Real mock sequencing table, F1-score percentage row; Genus is final three-column block, selecting kMS or Gtk rather than Species/Strain. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"TB2\", \"row_cells\": [\"F1-score %\", \"28.13\", \"85.42\", \"86.24\", \"87.18\", \"87.72\", \"95.83\", \"89.80\", \"94.85\"], \"selected_cell_zero_based\": 7, \"selected_cell_xml\": \"89.80\", \"caption\": \"Classification performance metrics measured on the real mock sequencing data.\"}","artifact_sha256":"4584e93ea035c1170b8756a0a52cbe99fe72e70bd09b5f1dee639ee104f78247","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11695915/fullTextXML"},"legacy_id":"lit-b3-030","legacy_row":{"id":"lit-b3-030","paper_id":"kmetashot-2025","domain_id":"microbes-communities","task":"Mock-community MAG taxonomy classification","model":"GTDB-Tk","model_version":"","dataset":"Real mock community MAGs","dataset_version":"","split":"","metric":"Genus-level F1","value":"89.80","unit":"%","uncertainty":"","protocol":"Genus classification of the same MAG set.","source_locator":"Table 2, F1-score % row, Genus Gtk column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11695915/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-031","kind":"result","name":"Lemur · F1 · Zymo LOG 10%","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-031"}],"attributes":{"printed_value":"0.376","numeric_value":"0.376","metric":"F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, LOG 10% / Lemur row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.420493+00:00","notes":"LOG 10% first block, F1 column. Kraken 2 row inherits dataset via rowspan; not LOG 75% or abundance Spearman. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"T3\", \"row_cells\": [\"LOG 10%\", \"Lemur\", \"0.500\", \"0.301\", \"0.376\", \"0.984\"], \"selected_cell_zero_based\": 4, \"selected_cell_xml\": \"0.376\", \"caption\": \"Mean performance and standard deviation across 5 replicate runs of all methods on Zymo LOG, bold values show best performance. Magnet does not report relative abundance, so the Spearman’s ρ cannot be computed. Tools listed below the horizontal dashed lines (for LOG 10% and LOG 75%) focus on the taxonomic classification of reads.\"}","artifact_sha256":"4afb9195da447916eb6f733816e3640741c7ade08ea8920d205c3be7b3cce27a","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11185576/fullTextXML"},"legacy_id":"lit-b3-031","legacy_row":{"id":"lit-b3-031","paper_id":"lemur-magnet-2024","domain_id":"microbes-communities","task":"Long-read taxonomic profiling","model":"Lemur","model_version":"","dataset":"Zymo LOG 10%","dataset_version":"","split":"","metric":"F1","value":"0.376","unit":"unitless","uncertainty":"","protocol":"Mean across five replicate runs on Zymo LOG 10%.","source_locator":"Table 3, LOG 10% / Lemur row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11185576/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-032","kind":"result","name":"Kraken 2 · F1 · Zymo LOG 10%","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-032"}],"attributes":{"printed_value":"0.375","numeric_value":"0.375","metric":"F1","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 3, LOG 10% / Kraken 2 row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.421888+00:00","notes":"LOG 10% first block, F1 column. Kraken 2 row inherits dataset via rowspan; not LOG 75% or abundance Spearman. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"T3\", \"row_cells\": [\"Kraken 2\", \"0.760\", \"0.249\", \"0.375\", \"0.910\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"0.375\", \"caption\": \"Mean performance and standard deviation across 5 replicate runs of all methods on Zymo LOG, bold values show best performance. Magnet does not report relative abundance, so the Spearman’s ρ cannot be computed. Tools listed below the horizontal dashed lines (for LOG 10% and LOG 75%) focus on the taxonomic classification of reads.\"}","artifact_sha256":"4afb9195da447916eb6f733816e3640741c7ade08ea8920d205c3be7b3cce27a","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11185576/fullTextXML"},"legacy_id":"lit-b3-032","legacy_row":{"id":"lit-b3-032","paper_id":"lemur-magnet-2024","domain_id":"microbes-communities","task":"Long-read taxonomic profiling","model":"Kraken 2","model_version":"","dataset":"Zymo LOG 10%","dataset_version":"","split":"","metric":"F1","value":"0.375","unit":"unitless","uncertainty":"","protocol":"Mean across five replicate runs on Zymo LOG 10%.","source_locator":"Table 3, LOG 10% / Kraken 2 row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11185576/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-033","kind":"result","name":"iPro-MP · Mean AUC · 23 independent prokaryotic promoter test sets","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-033"}],"attributes":{"printed_value":"0.935","numeric_value":"0.935","metric":"Mean AUC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, iPro-MP row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.361Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, iPro-MP row, AUC column; cell: 0.935","artifact_sha256":"d21541ee1f7a168da8e4a7c0f0e133c970cbe7bc41118f43a929f08b2fd2afd1","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12516880/fullTextXML"},"legacy_id":"lit-b3-033","legacy_row":{"id":"lit-b3-033","paper_id":"ipromp-2025","domain_id":"microbes-communities","task":"Multi-species prokaryotic promoter detection","model":"iPro-MP","model_version":"","dataset":"23 independent prokaryotic promoter test sets","dataset_version":"23 test sets","split":"independent test","metric":"Mean AUC","value":"0.935","unit":"unitless","uncertainty":"","protocol":"Average over independent testing sets.","source_locator":"Table 2, iPro-MP row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12516880/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-034","kind":"result","name":"Prompt · Mean AUC · 23 independent prokaryotic promoter test sets","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-034"}],"attributes":{"printed_value":"0.835","numeric_value":"0.835","metric":"Mean AUC","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, Prompt row, AUC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.361Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Prompt row, AUC column; cell: 0.835","artifact_sha256":"d21541ee1f7a168da8e4a7c0f0e133c970cbe7bc41118f43a929f08b2fd2afd1","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12516880/fullTextXML"},"legacy_id":"lit-b3-034","legacy_row":{"id":"lit-b3-034","paper_id":"ipromp-2025","domain_id":"microbes-communities","task":"Multi-species prokaryotic promoter detection","model":"Prompt","model_version":"","dataset":"23 independent prokaryotic promoter test sets","dataset_version":"23 test sets","split":"independent test","metric":"Mean AUC","value":"0.835","unit":"unitless","uncertainty":"","protocol":"Average over the same independent testing sets.","source_locator":"Table 2, Prompt row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12516880/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-035","kind":"result","name":"ICCTax · Genus macro AveP · ICCTax Complete dataset","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-035"}],"attributes":{"printed_value":"67.20","numeric_value":"67.20","metric":"Genus macro AveP","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, ICCTax row, Genus column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.373Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, ICCTax row, Genus column; cell: 67.20","artifact_sha256":"2ce0b48f1cde3aea7e561d92f4d7dc1525af7439ccd16f80bec0773e8812c8ec","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12619997/fullTextXML"},"legacy_id":"lit-b3-035","legacy_row":{"id":"lit-b3-035","paper_id":"icctax-2025","domain_id":"microbes-communities","task":"Hierarchical metagenomic taxonomy classification","model":"ICCTax","model_version":"","dataset":"ICCTax Complete dataset","dataset_version":"","split":"","metric":"Genus macro AveP","value":"67.20","unit":"%","uncertainty":"","protocol":"Macro average precision at genus rank on Complete dataset.","source_locator":"Table 2, ICCTax row, Genus column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12619997/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-036","kind":"result","name":"Kraken2 · Genus macro AveP · ICCTax Complete dataset","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-036"}],"attributes":{"printed_value":"70.56","numeric_value":"70.56","metric":"Genus macro AveP","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, Kraken2 row, Genus column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.373Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Kraken2 row, Genus column; cell: 70.56","artifact_sha256":"2ce0b48f1cde3aea7e561d92f4d7dc1525af7439ccd16f80bec0773e8812c8ec","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12619997/fullTextXML"},"legacy_id":"lit-b3-036","legacy_row":{"id":"lit-b3-036","paper_id":"icctax-2025","domain_id":"microbes-communities","task":"Hierarchical metagenomic taxonomy classification","model":"Kraken2","model_version":"","dataset":"ICCTax Complete dataset","dataset_version":"","split":"","metric":"Genus macro AveP","value":"70.56","unit":"%","uncertainty":"","protocol":"Macro average precision at genus rank on Complete dataset.","source_locator":"Table 2, Kraken2 row, Genus column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12619997/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-037","kind":"result","name":"Chai-1 · AUC-ROC · Antibody–antigen GEP test set","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-037"}],"attributes":{"printed_value":"0.86","numeric_value":"0.86","metric":"AUC-ROC","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.07","source_locator":"Table 5, Folded row, Chai-1 (no MSA) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.400Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, Folded row, Chai-1 (no MSA) column; cell: 0.86 ± 0.07","artifact_sha256":"57e64694c69052ed0495570e12ebfb4bb6c0ad152219f23827cd4b1cb53450ef","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12530544/fullTextXML"},"legacy_id":"lit-b3-037","legacy_row":{"id":"lit-b3-037","paper_id":"antibody-flexibility-2025","domain_id":"molecular-interactions","task":"Antibody–antigen interaction prediction using folded complexes","model":"Chai-1","model_version":"","dataset":"Antibody–antigen GEP test set","dataset_version":"","split":"","metric":"AUC-ROC","value":"0.86","unit":"unitless","uncertainty":"± 0.07","protocol":"Interaction classifier evaluated using Chai-1-folded input complexes; this is pipeline AUC, not DockQ.","source_locator":"Table 5, Folded row, Chai-1 (no MSA) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12530544/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-038","kind":"result","name":"Boltz-1 · AUC-ROC · Antibody–antigen GEP test set","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-038"}],"attributes":{"printed_value":"0.85","numeric_value":"0.85","metric":"AUC-ROC","metric_direction":"unknown","unit":"unitless","uncertainty":"± 0.05","source_locator":"Table 5, Folded row, Boltz-1 (no MSA) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.400Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 5, Folded row, Boltz-1 (no MSA) column; cell: 0.85 ± 0.05","artifact_sha256":"57e64694c69052ed0495570e12ebfb4bb6c0ad152219f23827cd4b1cb53450ef","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12530544/fullTextXML"},"legacy_id":"lit-b3-038","legacy_row":{"id":"lit-b3-038","paper_id":"antibody-flexibility-2025","domain_id":"molecular-interactions","task":"Antibody–antigen interaction prediction using folded complexes","model":"Boltz-1","model_version":"","dataset":"Antibody–antigen GEP test set","dataset_version":"","split":"","metric":"AUC-ROC","value":"0.85","unit":"unitless","uncertainty":"± 0.05","protocol":"Interaction classifier evaluated using Boltz-1-folded input complexes; this is pipeline AUC, not DockQ.","source_locator":"Table 5, Folded row, Boltz-1 (no MSA) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12530544/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-039","kind":"result","name":"Boltz-1 · Top-1 ligand RMSD <2 Å rate · Boltz-1 structure test set","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz1-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-039"}],"attributes":{"printed_value":"0.545","numeric_value":"0.545","metric":"Top-1 ligand RMSD <2 Å rate","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.424864+00:00","notes":"3 recycling rounds and 200 steps; L-RMSD <2 Angstrom top-1 (last column), not oracle. Five samples generated; top-1 means highest-confidence candidate. Repeated reference rows are one evaluation, not independent experiments. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"T4\", \"row_cells\": [\"3\", \"200\", \"0.729\", \"0.716\", \"0.654\", \"0.625\", \"0.621\", \"0.580\", \"0.581\", \"0.545\"], \"selected_cell_zero_based\": 9, \"selected_cell_xml\": \"0.545\", \"caption\": \"Ablation on the number of recycling rounds and sampling steps for Boltz-1 on the test set. We run the ablation study generating 5 samples and evaluating both the best (oracle) and highest confidence prediction (top-1) out of the 5 for every metric. All models used pre-computed MSAs with up to 4,096 sequences. It is worth noting that the metrics are noisy, so minor inconsistencies (e.g., lack of improvement with increased recycling rounds or diffusion steps) should not be overinterpreted. Moreover, there is a slight difference with the results in Figures 5 and 7 due to differences in MSA parameters as well as the set of structures passing all ablations.\"}","artifact_sha256":"1ebf712314d9a1c678ded989cc95a0c00c0331e5ad8c9f63194bc9780971d214","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11601547/fullTextXML"},"legacy_id":"lit-b3-039","legacy_row":{"id":"lit-b3-039","paper_id":"boltz1-2025","domain_id":"molecular-interactions","task":"Protein–ligand pose prediction","model":"Boltz-1","model_version":"3 recycling rounds; 200 diffusion steps","dataset":"Boltz-1 structure test set","dataset_version":"","split":"","metric":"Top-1 ligand RMSD <2 Å rate","value":"0.545","unit":"unitless","uncertainty":"","protocol":"Highest-confidence pose from five samples; precomputed MSAs up to 4,096 sequences.","source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11601547/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-040","kind":"result","name":"Ibex · Mean CDR H3 RMSD · ImmuneBuilder antibody test set","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-040"}],"attributes":{"printed_value":"2.72","numeric_value":"2.72","metric":"Mean CDR H3 RMSD","metric_direction":"unknown","unit":"Å","uncertainty":null,"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.426811+00:00","notes":"First Antibodies block, CDR H3 mean RMSD in Angstrom; excludes later Nanobodies and TCR blocks. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"t0001\", \"row_cells\": [\"\", \"Ibex\", \"0.61\", \"0.57\", \"2.72\", \"0.45\", \"0.57\", \"0.43\", \"0.98\", \"0.52\"], \"selected_cell_zero_based\": 4, \"selected_cell_xml\": \"2.72\", \"caption\": \"Average RMSD in angstrom, evaluated separately for each region on the ImmuneBuilder test set of antibodies, nanobodies and TCRs.\"}","artifact_sha256":"caa1109bd5fe7f6be703aa9d4afd6f4f1522bcbce6b7361650eb59618c2a9e14","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12710905/fullTextXML"},"legacy_id":"lit-b3-040","legacy_row":{"id":"lit-b3-040","paper_id":"ibex-2025","domain_id":"molecular-interactions","task":"Antibody loop structure prediction","model":"Ibex","model_version":"","dataset":"ImmuneBuilder antibody test set","dataset_version":"","split":"","metric":"Mean CDR H3 RMSD","value":"2.72","unit":"Å","uncertainty":"","protocol":"Backbone RMSD after framework alignment; average over antibody test structures.","source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12710905/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-041","kind":"result","name":"Chai-1 · Mean CDR H3 RMSD · ImmuneBuilder antibody test set","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-041"}],"attributes":{"printed_value":"2.65","numeric_value":"2.65","metric":"Mean CDR H3 RMSD","metric_direction":"unknown","unit":"Å","uncertainty":null,"source_locator":"Table 1, Antibodies / Chai-1 row, CDR H3 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.428536+00:00","notes":"First Antibodies block, CDR H3 mean RMSD in Angstrom; excludes later Nanobodies and TCR blocks. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"t0001\", \"row_cells\": [\"\", \"Chai-1\", \"0.67\", \"0.53\", \"2.65\", \"0.45\", \"0.53\", \"0.41\", \"1.20\", \"0.54\"], \"selected_cell_zero_based\": 4, \"selected_cell_xml\": \"2.65\", \"caption\": \"Average RMSD in angstrom, evaluated separately for each region on the ImmuneBuilder test set of antibodies, nanobodies and TCRs.\"}","artifact_sha256":"caa1109bd5fe7f6be703aa9d4afd6f4f1522bcbce6b7361650eb59618c2a9e14","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12710905/fullTextXML"},"legacy_id":"lit-b3-041","legacy_row":{"id":"lit-b3-041","paper_id":"ibex-2025","domain_id":"molecular-interactions","task":"Antibody loop structure prediction","model":"Chai-1","model_version":"","dataset":"ImmuneBuilder antibody test set","dataset_version":"","split":"","metric":"Mean CDR H3 RMSD","value":"2.65","unit":"Å","uncertainty":"","protocol":"Backbone RMSD after framework alignment; one seed and one diffusion trajectory.","source_locator":"Table 1, Antibodies / Chai-1 row, CDR H3 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12710905/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-042","kind":"result","name":"DEELIG · Pearson R · PDBbind core v2016","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-042"}],"attributes":{"printed_value":"0.889","numeric_value":"0.889","metric":"Pearson R","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, DEELIG row, PDBbind v2016 column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:55.586Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, DEELIG row, PDBbind v2016 column; cell: 0.889","artifact_sha256":"5a7620c18d0622561004e1e25b5cfaf7399e93df3547eeefdd4cf6d300bb8aba","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8274096/fullTextXML"},"legacy_id":"lit-b3-042","legacy_row":{"id":"lit-b3-042","paper_id":"deelig-2021","domain_id":"molecular-interactions","task":"Protein–ligand binding affinity prediction","model":"DEELIG","model_version":"","dataset":"PDBbind core v2016","dataset_version":"v2016","split":"","metric":"Pearson R","value":"0.889","unit":"unitless","uncertainty":"","protocol":"Source paper reports DEELIG on PDBbind core set.","source_locator":"Table 2, DEELIG row, PDBbind v2016 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8274096/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-043","kind":"result","name":"TOPBP (Complex) · Pearson R · PDBbind core v2016","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-043"}],"attributes":{"printed_value":"0.861","numeric_value":"0.861","metric":"Pearson R","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, TOPBP (Complex) row, PDBbind v2016 column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.429669+00:00","notes":"TOPBP Complex reference row; PDBbind v2016 core-set Pearson correlation. Third-party comparator with cited reference; do not infer an independent new run from table inclusion. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"table2-11779322211030364\", \"row_cells\": [\"TOPBP (Complex) 31\", \"0.808\", \"0.861\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"0.861\", \"caption\": \"Pearson correlations coefficient on PDBbind core set.\"}","artifact_sha256":"5a7620c18d0622561004e1e25b5cfaf7399e93df3547eeefdd4cf6d300bb8aba","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8274096/fullTextXML"},"legacy_id":"lit-b3-043","legacy_row":{"id":"lit-b3-043","paper_id":"deelig-2021","domain_id":"molecular-interactions","task":"Protein–ligand binding affinity prediction","model":"TOPBP (Complex)","model_version":"","dataset":"PDBbind core v2016","dataset_version":"v2016","split":"","metric":"Pearson R","value":"0.861","unit":"unitless","uncertainty":"","protocol":"Source table compiles a previously published comparator; protocol equivalence is not established.","source_locator":"Table 2, TOPBP (Complex) row, PDBbind v2016 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8274096/","evaluation_origin":"paper_compilation","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-044","kind":"result","name":"MolAS · RMSD ≤1 Å and PB-valid success · PoseBusters","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-044"}],"attributes":{"printed_value":"36.69","numeric_value":"36.69","metric":"RMSD ≤1 Å and PB-valid success","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.432565+00:00","notes":"PoseBusters, Mixed, AutoDock row within jointly trained with/without relaxation block. Selected RMSD <=1 Angstrom AND PB-valid group; five-fold average success percentage, not <=2 Angstrom. Inline bold digit nodes joined in original order. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"Tab3\", \"row_cells\": [\"PoseBusters\", \"Mixed\", \"AutoDock\", \"34.34\", \"36.69\", \"8.90\", \"51.17\", \"54.91\", \"11.87\"], \"selected_cell_zero_based\": 4, \"selected_cell_xml\": \"36.69\", \"caption\": \"Averaged 5-fold MolAS performance v.s. SBS across benchmarks\"}","artifact_sha256":"d556d47e0f7bbdc37eb62374b85ac9092dc7ff438fbae9e42892ac2cc784023f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13104262/fullTextXML"},"legacy_id":"lit-b3-044","legacy_row":{"id":"lit-b3-044","paper_id":"molas-2026","domain_id":"molecular-interactions","task":"Physically valid protein–ligand pose selection","model":"MolAS","model_version":"","dataset":"PoseBusters","dataset_version":"","split":"","metric":"RMSD ≤1 Å and PB-valid success","value":"36.69","unit":"%","uncertainty":"","protocol":"Averaged five-fold algorithm-selection performance on PoseBusters; joint RMSD and validity criterion.","source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13104262/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-045","kind":"result","name":"Single best solver · RMSD ≤1 Å and PB-valid success · PoseBusters","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-045"}],"attributes":{"printed_value":"34.34","numeric_value":"34.34","metric":"RMSD ≤1 Å and PB-valid success","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, SBS success column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.435454+00:00","notes":"PoseBusters, Mixed, AutoDock row within jointly trained with/without relaxation block. Selected RMSD <=1 Angstrom AND PB-valid group; five-fold average success percentage, not <=2 Angstrom. Inline bold digit nodes joined in original order. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"Tab3\", \"row_cells\": [\"PoseBusters\", \"Mixed\", \"AutoDock\", \"34.34\", \"36.69\", \"8.90\", \"51.17\", \"54.91\", \"11.87\"], \"selected_cell_zero_based\": 3, \"selected_cell_xml\": \"34.34\", \"caption\": \"Averaged 5-fold MolAS performance v.s. SBS across benchmarks\"}","artifact_sha256":"d556d47e0f7bbdc37eb62374b85ac9092dc7ff438fbae9e42892ac2cc784023f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13104262/fullTextXML"},"legacy_id":"lit-b3-045","legacy_row":{"id":"lit-b3-045","paper_id":"molas-2026","domain_id":"molecular-interactions","task":"Physically valid protein–ligand pose selection","model":"Single best solver","model_version":"","dataset":"PoseBusters","dataset_version":"","split":"","metric":"RMSD ≤1 Å and PB-valid success","value":"34.34","unit":"%","uncertainty":"","protocol":"Single best solver baseline under the same averaged five-fold selection test.","source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, SBS success column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13104262/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-046","kind":"result","name":"AutoDock Vina holo · Docked frames best-matched RMSD <3 Å · α-synuclein Ligand 47 MD ensemble","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-046"}],"attributes":{"printed_value":"27.96","numeric_value":"27.96","metric":"Docked frames best-matched RMSD <3 Å","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:56.275Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column; cell: 27.96 (98.30)","artifact_sha256":"d02d91cdde41cb76ec5c86b532dffc564879c69e764a8c6b7752460fbbfd24b7","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11785235/fullTextXML"},"legacy_id":"lit-b3-046","legacy_row":{"id":"lit-b3-046","paper_id":"ensemble-idp-docking-2025","domain_id":"molecular-interactions","task":"Intrinsically disordered protein ensemble docking","model":"AutoDock Vina holo","model_version":"","dataset":"α-synuclein Ligand 47 MD ensemble","dataset_version":"","split":"","metric":"Docked frames best-matched RMSD <3 Å","value":"27.96","unit":"%","uncertainty":"","protocol":"Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble.","source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11785235/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-047","kind":"result","name":"DiffDock holo · Docked frames best-matched RMSD <3 Å · α-synuclein Ligand 47 MD ensemble","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-047"}],"attributes":{"printed_value":"21.32","numeric_value":"21.32","metric":"Docked frames best-matched RMSD <3 Å","metric_direction":"unknown","unit":"%","uncertainty":null,"source_locator":"Table 2, Ligand 47 row, DiffDock Holo Docking column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:56.275Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Ligand 47 row, DiffDock Holo Docking column; cell: 21.32 (97.21)","artifact_sha256":"d02d91cdde41cb76ec5c86b532dffc564879c69e764a8c6b7752460fbbfd24b7","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11785235/fullTextXML"},"legacy_id":"lit-b3-047","legacy_row":{"id":"lit-b3-047","paper_id":"ensemble-idp-docking-2025","domain_id":"molecular-interactions","task":"Intrinsically disordered protein ensemble docking","model":"DiffDock holo","model_version":"","dataset":"α-synuclein Ligand 47 MD ensemble","dataset_version":"","split":"","metric":"Docked frames best-matched RMSD <3 Å","value":"21.32","unit":"%","uncertainty":"","protocol":"Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble.","source_locator":"Table 2, Ligand 47 row, DiffDock Holo Docking column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11785235/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-048","kind":"result","name":"AK-score-ensemble · Pearson R · CASF-2016","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-048"}],"attributes":{"printed_value":"0.812","numeric_value":"0.812","metric":"Pearson R","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.436853+00:00","notes":"CASF-2016 scoring Pearson R with learning rate0.0007. Single-model versus ensemble blocks kept distinct; ranking/docking scores not substituted. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"ijms-21-08424-t002\", \"row_cells\": [\"AK-score-ensemble\", \"0.0007\", \"0.812\", \"0.670\", \"0.589\", \"0.698\", \"36.0\", \"51.4\", \"59.7\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"\\n0.812\\n\", \"caption\": \"A comparison of prediction accuracy with the CASF-2016 dataset.\"}","artifact_sha256":"40cfd28dcd587599768ec99a6590ec593486475ff01c7b1d1f229b44aa91bf8d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7697539/fullTextXML"},"legacy_id":"lit-b3-048","legacy_row":{"id":"lit-b3-048","paper_id":"akscore-2020","domain_id":"molecular-interactions","task":"Protein–ligand binding affinity scoring","model":"AK-score-ensemble","model_version":"ensemble; learning rate 0.0007","dataset":"CASF-2016","dataset_version":"","split":"","metric":"Pearson R","value":"0.812","unit":"unitless","uncertainty":"","protocol":"CASF-2016 scoring-power evaluation.","source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7697539/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-049","kind":"result","name":"AK-score-single · Pearson R · CASF-2016","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-049"}],"attributes":{"printed_value":"0.759","numeric_value":"0.759","metric":"Pearson R","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.437894+00:00","notes":"CASF-2016 scoring Pearson R with learning rate0.0007. Single-model versus ensemble blocks kept distinct; ranking/docking scores not substituted. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"ijms-21-08424-t002\", \"row_cells\": [\"\", \"0.0007\", \"0.759\", \"0.616\", \"0.526\", \"0.640\", \"31.3\", \"47.1\", \"57.9\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"0.759\", \"caption\": \"A comparison of prediction accuracy with the CASF-2016 dataset.\"}","artifact_sha256":"40cfd28dcd587599768ec99a6590ec593486475ff01c7b1d1f229b44aa91bf8d","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC7697539/fullTextXML"},"legacy_id":"lit-b3-049","legacy_row":{"id":"lit-b3-049","paper_id":"akscore-2020","domain_id":"molecular-interactions","task":"Protein–ligand binding affinity scoring","model":"AK-score-single","model_version":"single; learning rate 0.0007","dataset":"CASF-2016","dataset_version":"","split":"","metric":"Pearson R","value":"0.759","unit":"unitless","uncertainty":"","protocol":"CASF-2016 scoring-power evaluation.","source_locator":"Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC7697539/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-050","kind":"result","name":"PMF + ECFP + PF (LightGBM) · Pearson R · Fingerprint-scoring benchmark","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-050"}],"attributes":{"printed_value":"0.79","numeric_value":"0.79","metric":"Pearson R","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.439460+00:00","notes":"Resolved two-row model rowspan: last LightGBM row is PMF+ECFP+PF; first LASSO row is PMF. Pearson R, not RMSE. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"tbl1\", \"row_cells\": [\"LightGBM\", \"0.79\", \"1.64\"], \"selected_cell_zero_based\": 1, \"selected_cell_xml\": \"0.79\", \"caption\": \"Pearson Correlation Coefficient (R) and RMSE between the Experimental Values and the Predicted Values by the Newly Developed Scoring Functions\"}","artifact_sha256":"47bd60c6392b801095fdb604de06c0d4bda6f555bae58e9955e491a5abf60576","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9178954/fullTextXML"},"legacy_id":"lit-b3-050","legacy_row":{"id":"lit-b3-050","paper_id":"fingerprint-scoring-2022","domain_id":"molecular-interactions","task":"Protein–ligand binding energy prediction","model":"PMF + ECFP + PF (LightGBM)","model_version":"","dataset":"Fingerprint-scoring benchmark","dataset_version":"","split":"","metric":"Pearson R","value":"0.79","unit":"unitless","uncertainty":"","protocol":"Binding-energy model using ligand and protein fingerprints with LightGBM.","source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9178954/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b3-051","kind":"result","name":"PMF (LASSO) · Pearson R · Fingerprint-scoring benchmark","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b3-051"}],"attributes":{"printed_value":"0.67","numeric_value":"0.67","metric":"Pearson R","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 1, PMF / LASSO row, R column","review":{"method":"independent_ai_table_review","reviewer":"Codex omics research agent; independent source-table review, not human review","reviewed_at":"2026-09-16T10:44:03.440812+00:00","notes":"Resolved two-row model rowspan: last LightGBM row is PMF+ECFP+PF; first LASSO row is PMF. Pearson R, not RMSE. Source check verifies central value and table context, not experiment reproduction or all metadata.","evidence":"{\"table_xml_id\": \"tbl1\", \"row_cells\": [\"PMF\", \"LASSO\", \"0.67\", \"2.04\"], \"selected_cell_zero_based\": 2, \"selected_cell_xml\": \"0.67\", \"caption\": \"Pearson Correlation Coefficient (R) and RMSE between the Experimental Values and the Predicted Values by the Newly Developed Scoring Functions\"}","artifact_sha256":"47bd60c6392b801095fdb604de06c0d4bda6f555bae58e9955e491a5abf60576","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9178954/fullTextXML"},"legacy_id":"lit-b3-051","legacy_row":{"id":"lit-b3-051","paper_id":"fingerprint-scoring-2022","domain_id":"molecular-interactions","task":"Protein–ligand binding energy prediction","model":"PMF (LASSO)","model_version":"","dataset":"Fingerprint-scoring benchmark","dataset_version":"","split":"","metric":"Pearson R","value":"0.67","unit":"unitless","uncertainty":"","protocol":"PMF-only LASSO baseline evaluated by the same authors.","source_locator":"Table 1, PMF / LASSO row, R column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9178954/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:29:32Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-001","kind":"result","name":"ARSENAL+ChromBPNet · AUROC · Yoruban LCL dsQTLs","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["regulatory-variant scoring"]},"source_ids":["arsenal-regulatory-dna-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-001"}],"attributes":{"printed_value":"0.896","numeric_value":"0.896","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":"±0.016","source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Yoruban LCL dsQTLs; ARSENAL+ChromBPNet AUROC Transcription verified; experimental claims not independently reproduced.","evidence":"0.896 ±0.016","artifact_sha256":"4a264956e47fc633aaff6573aac368dc691dd5de709b27c7421c078608ff542a","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12889687/fullTextXML"},"legacy_id":"lit-b4-001","legacy_row":{"id":"lit-b4-001","paper_id":"arsenal-regulatory-dna-2026","domain_id":"dna-genomes","task":"regulatory-variant scoring","model":"ARSENAL+ChromBPNet","model_version":"","dataset":"Yoruban LCL dsQTLs","dataset_version":"","split":"","metric":"AUROC","value":"0.896","unit":"fraction","uncertainty":"±0.016","protocol":"Supervised ChromBPNet variant scoring with ARSENAL motif-discovery regularization","source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12889687/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-002","kind":"result","name":"PlantCAD2 · AUROC · Andropogoneae genome-wide conservation","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["cross-species conservation prediction"]},"source_ids":["plantcad2-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-002"}],"attributes":{"printed_value":"0.725","numeric_value":"0.725","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"First comparison entry is PlantCAD2; AUROC is 0.725 versus comparator 0.691. Transcription verified; experimental claims not independently reproduced.","evidence":"0.725 vs 0.691","artifact_sha256":"4891955edb33af61e36dad51110574aa242e77c2df36429f100ff5a54bd8bd4b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12425018/fullTextXML"},"legacy_id":"lit-b4-002","legacy_row":{"id":"lit-b4-002","paper_id":"plantcad2-2025","domain_id":"dna-genomes","task":"cross-species conservation prediction","model":"PlantCAD2","model_version":"","dataset":"Andropogoneae genome-wide conservation","dataset_version":"","split":"","metric":"AUROC","value":"0.725","unit":"fraction","uncertainty":"","protocol":"Zero-shot score for conserved versus non-conserved sites from alignments of 35 Andropogoneae genomes","source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12425018/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-003","kind":"result","name":"Stacking-Auto · accuracy · enhancer independent comparison","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["hi-enhancer-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-003"}],"attributes":{"printed_value":"80.50","numeric_value":"80.50","metric":"accuracy","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Ours row, Accuracy column; original source method is the Stacking-Auto stage. Transcription verified; experimental claims not independently reproduced.","evidence":"80.50","artifact_sha256":"c86488c9f60329b7a3c4370598e7a0a9e4c8c45d1758b87007bfc8242376b009","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12758598/fullTextXML"},"legacy_id":"lit-b4-003","legacy_row":{"id":"lit-b4-003","paper_id":"hi-enhancer-2025","domain_id":"dna-genomes","task":"enhancer prediction","model":"Stacking-Auto","model_version":"","dataset":"enhancer independent comparison","dataset_version":"","split":"","metric":"accuracy","value":"80.50","unit":"percent","uncertainty":"","protocol":"Two-stage Hi-Enhancer system; paper Table 2 method comparison","source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12758598/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-004","kind":"result","name":"position-aware CNN · AUROC · human enhancer dataset","description":"","status":"source_checked","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["enhancer-position-encoding-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-004"}],"attributes":{"printed_value":"0.94","numeric_value":"0.94","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, Human section, CNN row, AUC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Human dataset row, CNN, AUC column. Transcription verified; experimental claims not independently reproduced.","evidence":"0.94","artifact_sha256":"0183b6a111b1b02344cad35a571a1fd2c56257e406c5be1df69f7902c5d06749","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11167433/fullTextXML"},"legacy_id":"lit-b4-004","legacy_row":{"id":"lit-b4-004","paper_id":"enhancer-position-encoding-2024","domain_id":"dna-genomes","task":"enhancer prediction","model":"position-aware CNN","model_version":"","dataset":"human enhancer dataset","dataset_version":"","split":"","metric":"AUROC","value":"0.94","unit":"fraction","uncertainty":"","protocol":"Nucleotide position-aware feature encoding; average assessment of CNN classifier","source_locator":"Table 2, Human section, CNN row, AUC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11167433/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-005","kind":"result","name":"ADAR-GPT continual · F1 · liver editing sites","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["A-to-I RNA editing site prediction"]},"source_ids":["adar-gpt-editing-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-005"}],"attributes":{"printed_value":"0.763","numeric_value":"0.763","metric":"F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, Adar-GPT (continual) row, F1 column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Adar-GPT continual row; XML inline decimal reordered by parser, original text verified separately. Transcription verified; experimental claims not independently reproduced.","evidence":"0.763","artifact_sha256":"cc8c7eb928f246f1f347a8822f614cd3475381c35eef6d579032ce441580198e","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12798952/fullTextXML"},"legacy_id":"lit-b4-005","legacy_row":{"id":"lit-b4-005","paper_id":"adar-gpt-editing-2026","domain_id":"rna-transcriptomes","task":"A-to-I RNA editing site prediction","model":"ADAR-GPT continual","model_version":"","dataset":"liver editing sites","dataset_version":"","split":"15% validation set","metric":"F1","value":"0.763","unit":"fraction","uncertainty":"","protocol":"Curriculum plus 15% fine-tuning; 201-nt sequence windows; decision threshold 0.5","source_locator":"Table 2, Adar-GPT (continual) row, F1 column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12798952/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-006","kind":"result","name":"R3Design · sequence recovery · Rfam","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA sequence design"]},"source_ids":["r3design-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-006"}],"attributes":{"printed_value":"43.27","numeric_value":"43.27","metric":"sequence recovery","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"R3Design row, first Recovery column Rfam; 43.27 plus/minus0.56. Transcription verified; experimental claims not independently reproduced.","evidence":"43.27 \\documentclass[12pt]{minimal} \\usepackage{amsmath} \\usepackage{wasysym} \\usepackage{amsfonts} \\usepackage{amssymb} \\usepackage{amsbsy} \\usepackage{upgreek} \\usepackage{mathrsfs} \\setlength{\\oddsidemargin}{-69pt} \\begin{document} $\\pm $\\end{document} 0.56","artifact_sha256":"b49c9ad846e46b11b240aace8e6bfaf953b842df166b69aee4843c02e9349779","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11685104/fullTextXML"},"legacy_id":"lit-b4-006","legacy_row":{"id":"lit-b4-006","paper_id":"r3design-2025","domain_id":"rna-transcriptomes","task":"RNA sequence design","model":"R3Design","model_version":"","dataset":"Rfam","dataset_version":"","split":"external","metric":"sequence recovery","value":"43.27","unit":"percent","uncertainty":"","protocol":"Tertiary-structure-conditioned RNA sequence design; external Rfam assessment","source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11685104/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-007","kind":"result","name":"CUPID Data-aug-Avg · AUROC · ncRNA interaction pairs","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["non-coding RNA pairwise interaction prediction"]},"source_ids":["cupid-rna-interactions-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-007"}],"attributes":{"printed_value":"0.919","numeric_value":"0.919","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"CUPID section Data-aug-Avg row; AUROC not AUPRC. Transcription verified; experimental claims not independently reproduced.","evidence":"0.919","artifact_sha256":"0e6719410b390ee9c4858bb9321042851100fb74df3aa109bf2af2b8aaff7ac1","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12957212/fullTextXML"},"legacy_id":"lit-b4-007","legacy_row":{"id":"lit-b4-007","paper_id":"cupid-rna-interactions-2026","domain_id":"rna-transcriptomes","task":"non-coding RNA pairwise interaction prediction","model":"CUPID Data-aug-Avg","model_version":"","dataset":"ncRNA interaction pairs","dataset_version":"","split":"","metric":"AUROC","value":"0.919","unit":"fraction","uncertainty":"","protocol":"Data augmentation with average pooling for molecule-level ncRNA embeddings","source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12957212/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-008","kind":"result","name":"ProteinBERT LLM-encoding model · AUROC · mRNA-RBP pairs","description":"","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA-protein interaction prediction"]},"source_ids":["mrna-protein-diversity-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-008"}],"attributes":{"printed_value":"71.5","numeric_value":"71.5","metric":"AUROC","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 2, RBP-aware test set row, auROC (%) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.257Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, RBP-aware test set row, auROC (%) column; cell: 71.5","artifact_sha256":"94f9fe22a5f0e6c8619e4af994eb4f6ded7417efcf0c1240380269280f97f9d1","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13235059/fullTextXML"},"legacy_id":"lit-b4-008","legacy_row":{"id":"lit-b4-008","paper_id":"mrna-protein-diversity-2026","domain_id":"rna-transcriptomes","task":"mRNA-protein interaction prediction","model":"ProteinBERT LLM-encoding model","model_version":"","dataset":"mRNA-RBP pairs","dataset_version":"","split":"RBP-aware test set","metric":"AUROC","value":"71.5","unit":"percent","uncertainty":"","protocol":"LLM encoding of protein partner; RBP-aware partition tests generalization to unseen protein diversity","source_locator":"Table 2, RBP-aware test set row, auROC (%) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13235059/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-009","kind":"result","name":"ESM2 650M · AUROC · human and viral proteins","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["human-versus-viral protein classification"]},"source_ids":["viral-immune-mimicry-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-009"}],"attributes":{"printed_value":"99.67","numeric_value":"99.67","metric":"AUROC","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 1, ESM2 650M row, AUC (%) column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.274Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 1, ESM2 650M row, AUC (%) column; cell: 99.67","artifact_sha256":"15250af2f75f70e2b6a3725d00bc7e276ed9d2d4a54f0ae9d6eaabf6be13e4a1","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12474240/fullTextXML"},"legacy_id":"lit-b4-009","legacy_row":{"id":"lit-b4-009","paper_id":"viral-immune-mimicry-2025","domain_id":"proteins-complexes","task":"human-versus-viral protein classification","model":"ESM2 650M","model_version":"","dataset":"human and viral proteins","dataset_version":"","split":"","metric":"AUROC","value":"99.67","unit":"percent","uncertainty":"","protocol":"ESM2 650M embedding-based human-virus classifier","source_locator":"Table 1, ESM2 650M row, AUC (%) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12474240/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-010","kind":"result","name":"ProtT5 embeddings + ensemble classifier · AUROC · Dset_448","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein binding-site prediction"]},"source_ids":["protein-binding-sites-2023"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-010"}],"attributes":{"printed_value":"0.810","numeric_value":"0.810","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Dset_448 block; ProtT5 AUROC, downstream ensemble retained in protocol. Transcription verified; experimental claims not independently reproduced.","evidence":"0.810","artifact_sha256":"491711aa7186e74bf33f6d601c4ea6a8f565e770938fe1f91a0cd347b8f06f3f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9849350/fullTextXML"},"legacy_id":"lit-b4-010","legacy_row":{"id":"lit-b4-010","paper_id":"protein-binding-sites-2023","domain_id":"proteins-complexes","task":"protein-protein binding-site prediction","model":"ProtT5 embeddings + ensemble classifier","model_version":"","dataset":"Dset_448","dataset_version":"","split":"","metric":"AUROC","value":"0.810","unit":"fraction","uncertainty":"","protocol":"Explainable ensemble binding-site predictor using ProtT5 features","source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9849350/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-011","kind":"result","name":"CLAPE-SMB with ESM-2 · AUROC · SJC","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["protein-small molecule binding-site prediction"]},"source_ids":["clape-smb-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-011"}],"attributes":{"printed_value":"0.917","numeric_value":"0.917","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 5, ESM-2 / SJC row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"ESM-2 on SJC AUROC. Transcription verified; experimental claims not independently reproduced.","evidence":"0.917","artifact_sha256":"215919244c3dd2dfb0b55fce91c211430fd8d4aee4bb28bd03eab9f4feb73e62","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11542454/fullTextXML"},"legacy_id":"lit-b4-011","legacy_row":{"id":"lit-b4-011","paper_id":"clape-smb-2024","domain_id":"proteins-complexes","task":"protein-small molecule binding-site prediction","model":"CLAPE-SMB with ESM-2","model_version":"","dataset":"SJC","dataset_version":"","split":"","metric":"AUROC","value":"0.917","unit":"fraction","uncertainty":"","protocol":"Contrastive CLAPE-SMB binding-site predictor with ESM-2 feature extractor","source_locator":"Table 5, ESM-2 / SJC row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11542454/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-012","kind":"result","name":"Vaxign-DL + ESM · AUPRC · vaccine candidate validation","description":"","status":"source_checked","facets":{"areas":["proteins-complexes"],"tasks":["vaccine-antigen candidate prediction"]},"source_ids":["vaxign-esm-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-012"}],"attributes":{"printed_value":"0.92","numeric_value":"0.92","metric":"AUPRC","metric_direction":"unknown","unit":"fraction","uncertainty":"±0.013","source_locator":"Table 2, 4 Layers row, AUPRC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Source spells 4 Layerss; AUPRC0.92±0.013. Transcription verified; experimental claims not independently reproduced.","evidence":"0.92 ± 0.013","artifact_sha256":"b76fff917addd0e9ff8a2fc843496132ecf832d3ceef248e3edf2dbc78baab5f","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11398487/fullTextXML"},"legacy_id":"lit-b4-012","legacy_row":{"id":"lit-b4-012","paper_id":"vaxign-esm-2024","domain_id":"proteins-complexes","task":"vaccine-antigen candidate prediction","model":"Vaxign-DL + ESM","model_version":"","dataset":"vaccine candidate validation","dataset_version":"","split":"","metric":"AUPRC","value":"0.92","unit":"fraction","uncertainty":"±0.013","protocol":"Combined skip architecture, four layers, ESM-generated sequence features","source_locator":"Table 2, 4 Layers row, AUPRC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11398487/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-013","kind":"result","name":"scGPT + residual geometry · AUROC · immune tissue","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["gene-regulatory signal prediction"]},"source_ids":["single-cell-residual-geometry-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-013"}],"attributes":{"printed_value":"0.677","numeric_value":"0.677","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Immune row, scGPT +geom (second numeric column), not Geneformer or delta. Transcription verified; experimental claims not independently reproduced.","evidence":"0.677","artifact_sha256":"77546faec51cfb5c78b73c5943f940c4e0097d130b1ddde4ab8287b507b36df6","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13418759/fullTextXML"},"legacy_id":"lit-b4-013","legacy_row":{"id":"lit-b4-013","paper_id":"single-cell-residual-geometry-2026","domain_id":"cells-tissues","task":"gene-regulatory signal prediction","model":"scGPT + residual geometry","model_version":"","dataset":"immune tissue","dataset_version":"","split":"","metric":"AUROC","value":"0.677","unit":"fraction","uncertainty":"","protocol":"Asymmetric extraction, PCA-64 centered cosine geometry added to scGPT baseline","source_locator":"Table 4, Immune row, scGPT > +geom AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13418759/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-014","kind":"result","name":"GREmLN · F1 · non-immune cells","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["cell-type annotation"]},"source_ids":["gremln-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-014"}],"attributes":{"printed_value":"0.937","numeric_value":"0.937","metric":"F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:57.502Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column; cell: 0.937","artifact_sha256":"3a20c4ededb749fc3f1120baf16dcfebe3fcb30418a91c445cfd91a7b5fdf553","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13060794/fullTextXML"},"legacy_id":"lit-b4-014","legacy_row":{"id":"lit-b4-014","paper_id":"gremln-2026","domain_id":"cells-tissues","task":"cell-type annotation","model":"GREmLN","model_version":"","dataset":"non-immune cells","dataset_version":"","split":"zero-shot","metric":"F1","value":"0.937","unit":"fraction","uncertainty":"","protocol":"Zero-shot cell-type annotation using pre-trained cellular graph foundation model","source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13060794/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-015","kind":"result","name":"Cell-DINO ViT-L · F1 · HPA-FoV","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["protein localization classification"]},"source_ids":["cell-dino-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-015"}],"attributes":{"printed_value":"65.5","numeric_value":"65.5","metric":"F1","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"HPA-FoV Cell-DINO PL column (protein localisation), not CL. Transcription verified; experimental claims not independently reproduced.","evidence":"65.5","artifact_sha256":"12a53a78c70b3033c3351cf7afd4da42ebc98bb3281308f07e71e5baffc153a0","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12826486/fullTextXML"},"legacy_id":"lit-b4-015","legacy_row":{"id":"lit-b4-015","paper_id":"cell-dino-2025","domain_id":"cells-tissues","task":"protein localization classification","model":"Cell-DINO ViT-L","model_version":"","dataset":"HPA-FoV","dataset_version":"","split":"","metric":"F1","value":"65.5","unit":"percent","uncertainty":"","protocol":"Self-supervised microscopy embedding pre-trained on HPA-FoV; downstream protein-localization classifier. Dataset-specific pretraining; the paper does not claim a general-purpose foundation model that generalizes beyond these benchmarks.","source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12826486/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-016","kind":"result","name":"scGen · precision at 50% recall · stimulated immune PBMC","description":"","status":"source_checked","facets":{"areas":["cells-tissues"],"tasks":["differentially expressed gene identification"]},"source_ids":["insilico-perturbation-auprc-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-016"}],"attributes":{"printed_value":"0.91","numeric_value":"0.91","metric":"precision at 50% recall","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"CD14+Mono scGen; precision at50%recall. Transcription verified; experimental claims not independently reproduced.","evidence":"0.91","artifact_sha256":"2715709d94f84744afa32cafdcaa72efd206d63af8c60afe7619b2cb90108b6b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12400816/fullTextXML"},"legacy_id":"lit-b4-016","legacy_row":{"id":"lit-b4-016","paper_id":"insilico-perturbation-auprc-2025","domain_id":"cells-tissues","task":"differentially expressed gene identification","model":"scGen","model_version":"","dataset":"stimulated immune PBMC","dataset_version":"","split":"CD14+Mono","metric":"precision at 50% recall","value":"0.91","unit":"fraction","uncertainty":"","protocol":"In-silico perturbation assessment with precision sampled at fixed 50% recall","source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12400816/","evaluation_origin":"independent_paper","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-017","kind":"result","name":"TCINet + HTRS · F1 · MetaHIT","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["pathogen detection"]},"source_ids":["metagenomic-pathogens-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-017"}],"attributes":{"printed_value":"0.84","numeric_value":"0.84","metric":"F1","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"MetaHIT block TCINet+HTRS F1-score. Transcription verified; experimental claims not independently reproduced.","evidence":"0.84","artifact_sha256":"aa88de1b0f0fd7ba1fedc1074bba9ba6ce199a0b723072a04d531db4585ae77c","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12493982/fullTextXML"},"legacy_id":"lit-b4-017","legacy_row":{"id":"lit-b4-017","paper_id":"metagenomic-pathogens-2025","domain_id":"microbes-communities","task":"pathogen detection","model":"TCINet + HTRS","model_version":"","dataset":"MetaHIT","dataset_version":"","split":"","metric":"F1","value":"0.84","unit":"fraction","uncertainty":"","protocol":"Taxonomy-constrained inference network with hierarchical taxonomy representation","source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12493982/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-018","kind":"result","name":"DETIRE · accuracy · testing viral metagenome dataset","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["viral sequence detection"]},"source_ids":["detire-viral-metagenomes-2023"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-018"}],"attributes":{"printed_value":"0.8772","numeric_value":"0.8772","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 1, Accuracy row, DETIRE column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.392Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 1, Accuracy row, DETIRE column; cell: 0.8772","artifact_sha256":"9ff7d32758620f7b0b0628425f62abff103ca2e33269ce3763383584bcebfc3c","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10313334/fullTextXML"},"legacy_id":"lit-b4-018","legacy_row":{"id":"lit-b4-018","paper_id":"detire-viral-metagenomes-2023","domain_id":"microbes-communities","task":"viral sequence detection","model":"DETIRE","model_version":"","dataset":"testing viral metagenome dataset","dataset_version":"","split":"test","metric":"accuracy","value":"0.8772","unit":"fraction","uncertainty":"","protocol":"Hybrid deep learning virus-fragment classifier on paper testing dataset","source_locator":"Table 1, Accuracy row, DETIRE column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10313334/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-019","kind":"result","name":"PC-mer + LR · accuracy · AMP","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["metagenomic genus classification"]},"source_ids":["pc-mer-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-019"}],"attributes":{"printed_value":"96.95","numeric_value":"96.95","metric":"accuracy","metric_direction":"unknown","unit":"percent","uncertainty":null,"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"AMP block PC-mer+LR section, k=8, first numeric value after k is Accuracy. Transcription verified; experimental claims not independently reproduced.","evidence":"96.95","artifact_sha256":"0b0a225fa6f5f3ba41dffc7b320c91f47738301cf45835260eecaa03b704a097","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11293629/fullTextXML"},"legacy_id":"lit-b4-019","legacy_row":{"id":"lit-b4-019","paper_id":"pc-mer-2024","domain_id":"microbes-communities","task":"metagenomic genus classification","model":"PC-mer + LR","model_version":"","dataset":"AMP","dataset_version":"","split":"genus-level","metric":"accuracy","value":"96.95","unit":"percent","uncertainty":"","protocol":"k=8 PC-mer feature extraction with logistic regression on AMP genus-classification dataset","source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11293629/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-020","kind":"result","name":"MDL4Microbiome · accuracy · CRC microbiome cohort","description":"","status":"source_checked","facets":{"areas":["microbes-communities"],"tasks":["microbiome disease-state classification"]},"source_ids":["mdl4microbiome-2022"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-020"}],"attributes":{"printed_value":"0.97","numeric_value":"0.97","metric":"accuracy","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 3, CRC row, MDL4Microbiome column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.492Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 3, CRC row, MDL4Microbiome column; cell: 0.97","artifact_sha256":"72330eda245ac97c5de5d47a491bb86d9b8f527a03ae25fa41cae5bfb637143b","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8763943/fullTextXML"},"legacy_id":"lit-b4-020","legacy_row":{"id":"lit-b4-020","paper_id":"mdl4microbiome-2022","domain_id":"microbes-communities","task":"microbiome disease-state classification","model":"MDL4Microbiome","model_version":"","dataset":"CRC microbiome cohort","dataset_version":"","split":"","metric":"accuracy","value":"0.97","unit":"fraction","uncertainty":"","protocol":"Multimodal deep learning model on colorectal-cancer versus healthy microbiome samples","source_locator":"Table 3, CRC row, MDL4Microbiome column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8763943/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-021","kind":"result","name":"binding-affinity meta-model · Pearson correlation · CASF-2016","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["protein-ligand binding affinity prediction"]},"source_ids":["ligand-affinity-meta-model-2024"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-021"}],"attributes":{"printed_value":"0.777","numeric_value":"0.777","metric":"Pearson correlation","metric_direction":"unknown","unit":"unitless","uncertainty":null,"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Meta-models CASF-2016 PCC. Confirmed XML training-set rowspan inherits preceding row, so0.777 maps to PCC. Transcription verified; experimental claims not independently reproduced.","evidence":"0.777","artifact_sha256":"0be25fe75bc0b2eb8065136555763bbae5964ea3a8fdb8c5de79ff96445f6a29","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11632770/fullTextXML"},"legacy_id":"lit-b4-021","legacy_row":{"id":"lit-b4-021","paper_id":"ligand-affinity-meta-model-2024","domain_id":"molecular-interactions","task":"protein-ligand binding affinity prediction","model":"binding-affinity meta-model","model_version":"","dataset":"CASF-2016","dataset_version":"","split":"core benchmark","metric":"Pearson correlation","value":"0.777","unit":"unitless","uncertainty":"","protocol":"Sequence-or-structure meta-model; predicts ln(Kd/Ki) using docked and deep-learning components","source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11632770/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-022","kind":"result","name":"DeepInterAware · AUROC · HIV neutralization","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["antigen-antibody HIV neutralization prediction"]},"source_ids":["deepinteraware-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-022"}],"attributes":{"printed_value":"0.826","numeric_value":"0.826","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":"±0.017","source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column","review":{"method":"independent_ai_table_review","reviewer":"Codex primary curator table review","reviewed_at":"2026-09-16T10:41:06Z","notes":"Ab Unseen block DeepInterAware AUROC0.826±0.017, not Ag Unseen. Transcription verified; experimental claims not independently reproduced.","evidence":"0.826 ± 0.017","artifact_sha256":"25d3561934965f754d8712ec02b2052e9a3979e433b88ecebd5db14e930f17a1","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11967782/fullTextXML"},"legacy_id":"lit-b4-022","legacy_row":{"id":"lit-b4-022","paper_id":"deepinteraware-2025","domain_id":"molecular-interactions","task":"antigen-antibody HIV neutralization prediction","model":"DeepInterAware","model_version":"","dataset":"HIV neutralization","dataset_version":"","split":"antibody-unseen","metric":"AUROC","value":"0.826","unit":"fraction","uncertainty":"±0.017","protocol":"Sequence-based interface-aware model, antibody-unseen split","source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11967782/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-023","kind":"result","name":"TransBind · AUROC · genome-wide TF binding sites","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["transcription-factor DNA binding-site prediction"]},"source_ids":["transbind-2026"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-023"}],"attributes":{"printed_value":"0.9508","numeric_value":"0.9508","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 2, TransBind row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.585Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 2, TransBind row, AUROC column; cell: 0.9508","artifact_sha256":"5d777f5925e941b7d087035d5d87e79ef75ae8d6456a770ffe8c527da566fee0","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13145115/fullTextXML"},"legacy_id":"lit-b4-023","legacy_row":{"id":"lit-b4-023","paper_id":"transbind-2026","domain_id":"molecular-interactions","task":"transcription-factor DNA binding-site prediction","model":"TransBind","model_version":"","dataset":"genome-wide TF binding sites","dataset_version":"","split":"test","metric":"AUROC","value":"0.9508","unit":"fraction","uncertainty":"","protocol":"Integrates protein and DNA embeddings for TFBS prediction on paper test dataset","source_locator":"Table 2, TransBind row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13145115/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"lit-b4-024","kind":"result","name":"ESM2_AMPS · AUROC · Bernett PPI dataset","description":"","status":"source_checked","facets":{"areas":["molecular-interactions"],"tasks":["protein-protein interaction prediction"]},"source_ids":["esm2-amp-2025"],"links":[{"relation":"evaluation","target_id":"evaluation-lit-b4-024"}],"attributes":{"printed_value":"0.68","numeric_value":"0.68","metric":"AUROC","metric_direction":"unknown","unit":"fraction","uncertainty":null,"source_locator":"Table 4, ESM2_AMPS row, AUROC column","review":{"method":"primary_xml_exact_label_cell_check","reviewer":"rewire deterministic table checker v1","reviewed_at":"2026-09-16T10:33:58.625Z","notes":"Exact row/header labels and numeric cell matched. Check verifies transcription, not experimental correctness.","evidence":"Table 4, ESM2_AMPS row, AUROC column; cell: 0.68","artifact_sha256":"8e7ad6efb72ca28d73037cdf465b0e62f99cd6d0ee4ca9eaf96a4c48da22fd6c","retrieval_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12392411/fullTextXML"},"legacy_id":"lit-b4-024","legacy_row":{"id":"lit-b4-024","paper_id":"esm2-amp-2025","domain_id":"molecular-interactions","task":"protein-protein interaction prediction","model":"ESM2_AMPS","model_version":"","dataset":"Bernett PPI dataset","dataset_version":"","split":"","metric":"AUROC","value":"0.68","unit":"fraction","uncertainty":"","protocol":"ESM2-derived embeddings plus paper interaction predictor","source_locator":"Table 4, ESM2_AMPS row, AUROC column","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12392411/","evaluation_origin":"author_reported","reviewed_utc":"2026-09-15T23:37:05Z"},"missing_metadata":{"model_version":"not_reported_in_legacy_extract","dataset_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","uncertainty":"not_reported_in_legacy_extract"}}} {"id":"mdl4microbiome-2022","kind":"source","name":"Multimodal deep learning applied to classify healthy and disease states of human microbiome","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8763943/","version":"PMC archival version PMC8763943.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1038/s41598-022-04773-3","publication_status":"peer_reviewed","year":2022,"artifact_sha256":"72330eda245ac97c5de5d47a491bb86d9b8f527a03ae25fa41cae5bfb637143b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8763943/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:58.492Z","legacy_paper":{"id":"mdl4microbiome-2022","title":"Multimodal deep learning applied to classify healthy and disease states of human microbiome","year":2022,"publication_status":"peer_reviewed","version":"PMC archival version PMC8763943.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8763943/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Scientific Reports; PMC ID: PMC8763943. Study predates most microbial foundation models; useful task baseline only.","doi":"10.1038/s41598-022-04773-3"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"megsite-2025","kind":"source","name":"MegSite: an accurate nucleic acid-binding residue prediction method based on multimodal protein language model","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12496013/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bib/bbaf524","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"10d13122331813243d83b84fe6f9294eac7e7c03cde082ebed276191ac41089c","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12496013/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558213+00:00","legacy_paper":{"id":"megsite-2025","title":"MegSite: an accurate nucleic acid-binding residue prediction method based on multimodal protein language model","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12496013/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bib/bbaf524","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Briefings in Bioinformatics."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"metagenomic-pathogens-2025","kind":"source","name":"Enhancing pathogen identification through AI-assisted metagenomic sequencing","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12493982/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.3389/fmicb.2025.1634194","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"aa88de1b0f0fd7ba1fedc1074bba9ba6ce199a0b723072a04d531db4585ae77c","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12493982/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"metagenomic-pathogens-2025","title":"Enhancing pathogen identification through AI-assisted metagenomic sequencing","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12493982/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Frontiers in Microbiology; PMC ID: PMC12493982. Primary article has mixed biomedical-text and metagenomics assessments; selected MetaHIT pathogen-detection table only.","doi":"10.3389/fmicb.2025.1634194"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"molas-2026","kind":"source","name":"Molecular embedding-based algorithm selection in protein-ligand docking","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13104262/","version":"PMC archival version PMC13104262.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1186/s13321-026-01168-8","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"d556d47e0f7bbdc37eb62374b85ac9092dc7ff438fbae9e42892ac2cc784023f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13104262/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.432565+00:00","legacy_paper":{"id":"molas-2026","title":"Molecular embedding-based algorithm selection in protein-ligand docking","year":2026,"publication_status":"peer_reviewed","version":"PMC archival version PMC13104262.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13104262/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Journal of Cheminformatics; PMC ID: PMC13104262.","doi":"10.1186/s13321-026-01168-8"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"mouse-geneformer-2025","kind":"source","name":"Mouse-Geneformer: A deep learning model for mouse single-cell transcriptome and its cross-species utility","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11964219/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1371/journal.pgen.1011420","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"ef6c68f5c9b47c2f89598ddf647f05b72e4155e8b33609ebff23936a84bbd41d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11964219/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.392488+00:00","legacy_paper":{"id":"mouse-geneformer-2025","title":"Mouse-Geneformer: A deep learning model for mouse single-cell transcriptome and its cross-species utility","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11964219/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: PLOS Genetics; PMC ID: PMC11964219.","doi":"10.1371/journal.pgen.1011420"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"mpro-pose-affinity-2025","kind":"source","name":"A Comparative Study of Deep Learning and Classical Modeling Approaches for Protein–Ligand Binding Pose and Affinity Prediction in Coronavirus Main Proteases","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12801289/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acs.jcim.5c02481","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"c356a1c65a0033e5ae18a05d4afab5495856c5b6869328ff49e13547a4801a57","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12801289/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.557756+00:00","legacy_paper":{"id":"mpro-pose-affinity-2025","title":"A Comparative Study of Deep Learning and Classical Modeling Approaches for Protein–Ligand Binding Pose and Affinity Prediction in Coronavirus Main Proteases","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12801289/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC12801289.","doi":"10.1021/acs.jcim.5c02481"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"mrna-lm-2025","kind":"source","name":"mRNA-LM: full-length integrated SLM for mRNA analysis","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11962594/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/nar/gkaf044","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"3a23de3c672ec162d13561c483f180a73b550d717256deffdc9099accec205fd","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11962594/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558214+00:00","legacy_paper":{"id":"mrna-lm-2025","title":"mRNA-LM: full-length integrated SLM for mRNA analysis","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11962594/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/nar/gkaf044","notes":"Numeric result checked against Table 1. in primary full-text XML; journal/source: Nucleic Acids Research."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"mrna-protein-diversity-2026","kind":"source","name":"Generalizable deep-learning-based mRNA-protein interaction prediction strongly depends on protein diversity","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13235059/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1186/s13321-026-01197-3","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"94f9fe22a5f0e6c8619e4af994eb4f6ded7417efcf0c1240380269280f97f9d1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13235059/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:57.257Z","legacy_paper":{"id":"mrna-protein-diversity-2026","title":"Generalizable deep-learning-based mRNA-protein interaction prediction strongly depends on protein diversity","year":2026,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13235059/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Journal of Cheminformatics; PMC ID: PMC13235059. ProteinBERT encodes the protein side of an mRNA-protein task; score is not an RNA foundation-model result.","doi":"10.1186/s13321-026-01197-3"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"mrnabench-2025","kind":"source","name":"mRNABench: A curated benchmark for mature mRNA property and function prediction","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12265608/","version":"preprint archived 2025-07-08","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2025.07.05.662870","publication_status":"preprint","year":2025,"artifact_sha256":"79f6264ee883535203c63a313547e7c57baa85585f76b42f8d899eb17fb7e600","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12265608/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.497221+00:00","legacy_paper":{"id":"mrnabench-2025","title":"mRNABench: A curated benchmark for mature mRNA property and function prediction","year":2025,"publication_status":"preprint","version":"preprint archived 2025-07-08","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12265608/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC12265608.","doi":"10.1101/2025.07.05.662870"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"mrnabert-2025","kind":"source","name":"mRNABERT: advancing mRNA sequence design with a universal language model and comprehensive dataset","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12644827/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s41467-025-65340-8","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"ff08ba895b7080446c08a930548b48a0041ae990c222ebb07e6ba7dcaf48ad44","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12644827/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558216+00:00","legacy_paper":{"id":"mrnabert-2025","title":"mRNABERT: advancing mRNA sequence design with a universal language model and comprehensive dataset","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12644827/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s41467-025-65340-8","notes":"Numeric result checked against Table 2 in primary full-text XML; journal/source: Nature Communications."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"mulan-2025","kind":"source","name":"MULAN: multimodal protein language model for sequence and structure encoding","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12452268/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioadv/vbaf117","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"771a9a26ebda6f49ea266540e8dd6e6de0cbaef724de818ca6124a5f9c50d350","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12452268/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558217+00:00","legacy_paper":{"id":"mulan-2025","title":"MULAN: multimodal protein language model for sequence and structure encoding","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12452268/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioadv/vbaf117","notes":"Numeric result checked against Table 2. in primary full-text XML; journal/source: Bioinformatics Advances."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"nabas-plus-2025","kind":"source","name":"Advancing metagenomic classification with NABAS+: a novel alignment-based approach","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12231603/","version":"PMC12231603.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1093/nargab/lqaf092","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"49903e751beb86f6744825d2fdb3ea2fbe327b52b1ce68c323bfa8b66dae71ec","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12231603/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.546107+00:00","legacy_paper":{"id":"nabas-plus-2025","title":"Advancing metagenomic classification with NABAS+: a novel alignment-based approach","year":2025,"publication_status":"peer_reviewed","version":"PMC12231603.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12231603/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: NAR Genomics and Bioinformatics; PMC ID: PMC12231603.","doi":"10.1093/nargab/lqaf092"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"ncd-metagenomics-2026","kind":"source","name":"Normalized compression distance for DNA classification","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12884959/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.7717/peerj.20677","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"10e9ba780c45e7787baff9b81ef7b45c014d7fe6c716d6c759e14d89a813dc1c","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12884959/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.414806+00:00","legacy_paper":{"id":"ncd-metagenomics-2026","title":"Normalized compression distance for DNA classification","year":2026,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12884959/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: PeerJ; PMC ID: PMC12884959.","doi":"10.7717/peerj.20677"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"nmdn-2025","kind":"source","name":"Normalized Protein–Ligand Distance Likelihood Score for End-to-End Blind Docking and Virtual Screening","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11815853/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1021/acs.jcim.4c01014","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"194b21478aaedd9a7384cabb8b0040ca5b6a4938f4d275627b86a6b787affc20","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11815853/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.552697+00:00","legacy_paper":{"id":"nmdn-2025","title":"Normalized Protein–Ligand Distance Likelihood Score for End-to-End Blind Docking and Virtual Screening","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11815853/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Journal of Chemical Information and Modeling; PMC ID: PMC11815853.","doi":"10.1021/acs.jcim.4c01014"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"pc-mer-2024","kind":"source","name":"PC-mer: An Ultra-fast memory-efficient tool for metagenomics profiling and classification","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11293629/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1371/journal.pone.0307279","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"0b0a225fa6f5f3ba41dffc7b320c91f47738301cf45835260eecaa03b704a097","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11293629/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"pc-mer-2024","title":"PC-mer: An Ultra-fast memory-efficient tool for metagenomics profiling and classification","year":2024,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11293629/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: PLOS ONE; PMC ID: PMC11293629. Feature-extraction classifier; not a biological foundation model.","doi":"10.1371/journal.pone.0307279"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"phylogpn-2025","kind":"source","name":"A Phylogenetic Approach to Genomic Language Modeling","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11908359/","version":"preprint version in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":null,"publication_status":"preprint","year":2025,"artifact_sha256":"807f3a26cbfa9b5ce238d92164bd523302c67d1c5794b08273c51cca1acd4224","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11908359/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558218+00:00","legacy_paper":{"id":"phylogpn-2025","title":"A Phylogenetic Approach to Genomic Language Modeling","year":2025,"publication_status":"preprint","version":"preprint version in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11908359/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","notes":"Numeric result checked against Table 1. in primary full-text XML; journal/source: ArXiv."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"plantcad2-2025","kind":"source","name":"PlantCAD2: A Long-Context DNA Language Model for Cross-Species Functional Annotation in Angiosperms","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12425018/","version":"preprint version in PMC","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1101/2025.08.27.672609","publication_status":"preprint","year":2025,"artifact_sha256":"4891955edb33af61e36dad51110574aa242e77c2df36429f100ff5a54bd8bd4b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12425018/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"plantcad2-2025","title":"PlantCAD2: A Long-Context DNA Language Model for Cross-Species Functional Annotation in Angiosperms","year":2025,"publication_status":"preprint","version":"preprint version in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12425018/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: bioRxiv; PMC ID: PMC12425018. Preprint; Table 1 pairs PlantCAD2 with an unnamed best benchmark; only PlantCAD2 value recorded.","doi":"10.1101/2025.08.27.672609"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"polya-glm-2025","kind":"source","name":"PolyA-GLM: A comprehensive framework for De novo polyadenylation site prediction using genome language models","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12799945/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1016/j.csbj.2025.12.011","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"e9ebd53d88837ad8d457881ffee918d2734dcae87d3c5cd03135947b6cf5dbde","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12799945/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558220+00:00","legacy_paper":{"id":"polya-glm-2025","title":"PolyA-GLM: A comprehensive framework for De novo polyadenylation site prediction using genome language models","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12799945/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1016/j.csbj.2025.12.011","notes":"Numeric result checked against Table 1 in primary full-text XML; journal/source: Computational and Structural Biotechnology Journal."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"prime-2026","kind":"source","name":"PRIME: An evaluation framework for protein representation inference and generalization in viral mutation space","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13425921/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1186/s12864-026-12976-5","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"f6aac4c25dd93026f87ce9a2e327c95faf4c3014d7f9ae04bb11f208ce047971","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13425921/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.525183+00:00","legacy_paper":{"id":"prime-2026","title":"PRIME: An evaluation framework for protein representation inference and generalization in viral mutation space","year":2026,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13425921/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: BMC Genomics; PMC ID: PMC13425921.","doi":"10.1186/s12864-026-12976-5"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"profile-association-0bf5ead3625ab59bbc49","kind":"claim","name":"Open Problems: evaluates task Batch integration","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["single-cell"]},"source_ids":["src-discovery-openproblems-bio-openproblems"],"links":[{"relation":"subject","target_id":"discovery-benchmark-open-problems"}],"attributes":{"field":"links:evaluates_task:catalog-task-cell-batch-integration","target_id":"catalog-task-cell-batch-integration","source_locator":"Official linked benchmark directory, Batch Integration entry: https://openproblems.bio/benchmarks/; platform task membership, not protocol equivalence","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-0d52555df24c3c1f90f6","kind":"claim","name":"RNA-FM: family RNA-FM","description":"Both records point to the same official project; the catalogue record has no independently identified checkpoint. Preserve both IDs and restrict shared explanation to family-level facts.","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"method_types":["foundation model"]},"source_ids":["src-discovery-ml4bio-rna-fm"],"links":[{"relation":"subject","target_id":"catalog-model-rna-fm"}],"attributes":{"field":"links:family:discovery-model-rna-fm","target_id":"discovery-model-rna-fm","source_locator":"Official README project identity and the existing catalogue source URL","review":{"method":"automated_source_review","date":"2026-09-16","note":"Both records point to the same official project; the catalogue record has no independently identified checkpoint. Preserve both IDs and restrict shared explanation to family-level facts."}}} {"id":"profile-association-0fb2930451b33229a2e0","kind":"claim","name":"GEARS: family GEARS","description":"Both records point to the same official project; the catalogue record has no independently identified checkpoint. Preserve both IDs and restrict shared explanation to family-level facts.","status":"source_checked","facets":{"areas":["cells-tissues"],"method_types":["specialist"]},"source_ids":["src-discovery-snap-stanford-gears"],"links":[{"relation":"subject","target_id":"catalog-model-gears"}],"attributes":{"field":"links:family:discovery-model-gears","target_id":"discovery-model-gears","source_locator":"Official README project identity and the existing catalogue source URL","review":{"method":"automated_source_review","date":"2026-09-16","note":"Both records point to the same official project; the catalogue record has no independently identified checkpoint. Preserve both IDs and restrict shared explanation to family-level facts."}}} {"id":"profile-association-10c89dc6780906fcd514","kind":"claim","name":"SpliceAI: variant of SpliceAI","description":"The official pinned package version matches the named catalogue version. This does not equate every annotation or ensemble run configuration.","status":"source_checked","facets":{"areas":["dna-genomes"],"method_types":["specialist"]},"source_ids":["src-discovery-illumina-spliceai"],"links":[{"relation":"subject","target_id":"catalog-model-spliceai"}],"attributes":{"field":"links:variant_of:discovery-model-spliceai","target_id":"discovery-model-spliceai","source_locator":"setup.py line 9 at 03f42437aaf56dc5dfd822c4ccee5aec1a705079 declares version 1.3.1; README identifies SpliceAI","review":{"method":"automated_source_review","date":"2026-09-16","note":"The official pinned package version matches the named catalogue version. This does not equate every annotation or ensemble run configuration."}}} {"id":"profile-association-10ecbaabf760dbba79c8","kind":"claim","name":"GlycanML taxonomy prediction: part of GlycanML","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["glycomics"]},"source_ids":["src-discovery-glycanml-glycanml"],"links":[{"relation":"subject","target_id":"discovery-benchmark-glycanml-taxonomy-prediction"}],"attributes":{"field":"links:part_of:discovery-benchmark-glycanml","target_id":"discovery-benchmark-glycanml","source_locator":"README: Overview, task list","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-1d86b58d85bc57cbe366","kind":"claim","name":"mRNA-FM: variant of RNA-FM","description":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities.","status":"source_checked","facets":{"areas":["rna-transcriptomes"],"method_types":["foundation model"]},"source_ids":["src-discovery-ml4bio-rna-fm"],"links":[{"relation":"subject","target_id":"catalog-model-mrna-fm"}],"attributes":{"field":"links:variant_of:discovery-model-rna-fm","target_id":"discovery-model-rna-fm","source_locator":"README.md: introduction identifies mRNA-FM as the coding-sequence extension of RNA-FM","review":{"method":"automated_source_review","date":"2026-09-16","note":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities."}}} {"id":"profile-association-1e3a9f8e48d3e9042bca","kind":"claim","name":"ESMFold: variant of ESMFold","description":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities.","status":"source_checked","facets":{"areas":["proteins-complexes"],"method_types":["foundation model"]},"source_ids":["src-discovery-facebookresearch-esm"],"links":[{"relation":"subject","target_id":"catalog-model-esmfold"}],"attributes":{"field":"links:variant_of:discovery-model-esmfold","target_id":"discovery-model-esmfold","source_locator":"README.md: ESMFold Structure Prediction distinguishes v0 and v1","review":{"method":"automated_source_review","date":"2026-09-16","note":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities."}}} {"id":"profile-association-2fb39a063eed368e689b","kind":"claim","name":"Boltz-2: variant of Boltz","description":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities.","status":"source_checked","facets":{"areas":["molecular-interactions"],"method_types":["foundation model"]},"source_ids":["src-discovery-jwohlwend-boltz"],"links":[{"relation":"subject","target_id":"catalog-model-boltz-2"}],"attributes":{"field":"links:variant_of:discovery-model-boltz","target_id":"discovery-model-boltz","source_locator":"README.md: Introduction distinguishes Boltz-1 and Boltz-2 in the model family","review":{"method":"automated_source_review","date":"2026-09-16","note":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities."}}} {"id":"profile-association-3242cc83bee01ceb529e","kind":"claim","name":"GlycanML protein-glycan interaction prediction: part of GlycanML","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["glycomics"]},"source_ids":["src-discovery-glycanml-glycanml"],"links":[{"relation":"subject","target_id":"discovery-benchmark-glycanml-protein-glycan-interaction-prediction"}],"attributes":{"field":"links:part_of:discovery-benchmark-glycanml","target_id":"discovery-benchmark-glycanml","source_locator":"README: Overview, task list","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-3fc720d75e5875470593","kind":"claim","name":"SpliceAI 1.3.1: uses model SpliceAI","description":"The evaluated MFASS pipeline explicitly uses the named base model. Adaptation, annotation and masking remain properties of the pipeline, not the family.","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source","src-discovery-illumina-spliceai"],"links":[{"relation":"subject","target_id":"rewire-model-spliceai-1-3-1"}],"attributes":{"field":"links:uses_model:discovery-model-spliceai","target_id":"discovery-model-spliceai","source_locator":"Pinned rewire benchmarks/mfass/README.md and corresponding results JSON, config; official model README","review":{"method":"automated_source_review","date":"2026-09-16","note":"The evaluated MFASS pipeline explicitly uses the named base model. Adaptation, annotation and masking remain properties of the pipeline, not the family."}}} {"id":"profile-association-49f1b5328fff0bc440af","kind":"claim","name":"MFASS v2: evaluates task MFASS splice-variant prioritisation","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"subject","target_id":"rewire-mfass-v2"}],"attributes":{"field":"links:evaluates_task:catalog-task-mfass-splice","target_id":"catalog-task-mfass-splice","source_locator":"Pinned benchmarks/mfass/README.md: Dataset; Cohort reconciliation; Split","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-4aeaae5057842d41e084","kind":"claim","name":"Pangolin · mask=False: uses model Pangolin","description":"The evaluated MFASS pipeline explicitly uses the named base model. Adaptation, annotation and masking remain properties of the pipeline, not the family.","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source","src-discovery-tkzeng-pangolin"],"links":[{"relation":"subject","target_id":"rewire-model-pangolin-maskfalse"}],"attributes":{"field":"links:uses_model:discovery-model-pangolin","target_id":"discovery-model-pangolin","source_locator":"Pinned rewire benchmarks/mfass/README.md and corresponding results JSON, config; official model README","review":{"method":"automated_source_review","date":"2026-09-16","note":"The evaluated MFASS pipeline explicitly uses the named base model. Adaptation, annotation and masking remain properties of the pipeline, not the family."}}} {"id":"profile-association-507834867ec7cde103f9","kind":"claim","name":"scGPT: variant of scGPT","description":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities.","status":"source_checked","facets":{"areas":["cells-tissues"],"method_types":["foundation model"]},"source_ids":["src-discovery-bowang-lab-scgpt"],"links":[{"relation":"subject","target_id":"catalog-model-scgpt"}],"attributes":{"field":"links:variant_of:discovery-model-scgpt","target_id":"discovery-model-scgpt","source_locator":"README.md: Pretrained scGPT checkpoints, whole-human row","review":{"method":"automated_source_review","date":"2026-09-16","note":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities."}}} {"id":"profile-association-524d4e1d2f3954f416b3","kind":"claim","name":"GlycanML glycosylation type prediction: part of GlycanML","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["glycomics"]},"source_ids":["src-discovery-glycanml-glycanml"],"links":[{"relation":"subject","target_id":"discovery-benchmark-glycanml-glycosylation-type-prediction"}],"attributes":{"field":"links:part_of:discovery-benchmark-glycanml","target_id":"discovery-benchmark-glycanml","source_locator":"README: Overview, task list","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-56fc7ee6772f13885b95","kind":"claim","name":"DNABERT-2 117M · frozen pair embeddings: uses model DNABERT-2","description":"The evaluated MFASS pipeline explicitly uses the named base model. Adaptation, annotation and masking remain properties of the pipeline, not the family.","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source","src-discovery-magics-lab-dnabert-2"],"links":[{"relation":"subject","target_id":"rewire-model-dnabert2-117m-frozen-pair-logreg"}],"attributes":{"field":"links:uses_model:discovery-model-dnabert-2","target_id":"discovery-model-dnabert-2","source_locator":"Pinned rewire benchmarks/mfass/README.md and corresponding results JSON, config; official model README","review":{"method":"automated_source_review","date":"2026-09-16","note":"The evaluated MFASS pipeline explicitly uses the named base model. Adaptation, annotation and masking remain properties of the pipeline, not the family."}}} {"id":"profile-association-5dc7c010a32600e0fbd1","kind":"claim","name":"CAMI metagenome assembly: part of CAMI","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["microbiome"]},"source_ids":["src-discovery-cami"],"links":[{"relation":"subject","target_id":"discovery-benchmark-cami-metagenome-assembly"}],"attributes":{"field":"links:part_of:discovery-benchmark-cami","target_id":"discovery-benchmark-cami","source_locator":"CAMI official home page: Summary / Per category","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-6f91ef59fe79ba4e85e8","kind":"claim","name":"ProteinMPNN: variant of ProteinMPNN","description":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities.","status":"source_checked","facets":{"areas":["proteins-complexes"],"method_types":["specialist"]},"source_ids":["src-discovery-dauparas-proteinmpnn"],"links":[{"relation":"subject","target_id":"catalog-model-proteinmpnn"}],"attributes":{"field":"links:variant_of:discovery-model-proteinmpnn","target_id":"discovery-model-proteinmpnn","source_locator":"README.md: full protein backbone model list includes v_48_020","review":{"method":"automated_source_review","date":"2026-09-16","note":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities."}}} {"id":"profile-association-7248dbde8b4a47c0358d","kind":"claim","name":"TAPE Contact Prediction: part of TAPE","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["protein-function"]},"source_ids":["src-discovery-songlab-cal-tape"],"links":[{"relation":"subject","target_id":"discovery-benchmark-tape-contact-prediction"}],"attributes":{"field":"links:part_of:discovery-benchmark-tape","target_id":"discovery-benchmark-tape","source_locator":"README: List of Models and Tasks; Leaderboard","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-7620c8aef622f0317b76","kind":"claim","name":"MassSpecGym Molecule retrieval: part of MassSpecGym","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["metabolomics"]},"source_ids":["src-discovery-pluskal-lab-massspecgym"],"links":[{"relation":"subject","target_id":"discovery-benchmark-massspecgym-molecule-retrieval"}],"attributes":{"field":"links:part_of:discovery-benchmark-massspecgym","target_id":"discovery-benchmark-massspecgym","source_locator":"README: opening challenge list","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-874580c7cba4c198c076","kind":"claim","name":"scIB: evaluates task Batch integration","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["single-cell"]},"source_ids":["src-discovery-theislab-scib"],"links":[{"relation":"subject","target_id":"discovery-benchmark-scib"}],"attributes":{"field":"links:evaluates_task:catalog-task-cell-batch-integration","target_id":"catalog-task-cell-batch-integration","source_locator":"README: Metrics, Biological Conservation and Batch Correction","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-923959d8f16ef46a494d","kind":"claim","name":"TAPE Secondary Structure: part of TAPE","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["protein-function"]},"source_ids":["src-discovery-songlab-cal-tape"],"links":[{"relation":"subject","target_id":"discovery-benchmark-tape-secondary-structure"}],"attributes":{"field":"links:part_of:discovery-benchmark-tape","target_id":"discovery-benchmark-tape","source_locator":"README: List of Models and Tasks; Leaderboard","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-9c3fcc0e37012e7dcbbd","kind":"claim","name":"MassSpecGym De novo molecule generation: part of MassSpecGym","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["metabolomics"]},"source_ids":["src-discovery-pluskal-lab-massspecgym"],"links":[{"relation":"subject","target_id":"discovery-benchmark-massspecgym-de-novo-molecule-generation"}],"attributes":{"field":"links:part_of:discovery-benchmark-massspecgym","target_id":"discovery-benchmark-massspecgym","source_locator":"README: opening challenge list","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-9ee0416aed06f5b6aa75","kind":"claim","name":"TAPE Remote Homology Detection: part of TAPE","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["protein-function"]},"source_ids":["src-discovery-songlab-cal-tape"],"links":[{"relation":"subject","target_id":"discovery-benchmark-tape-remote-homology-detection"}],"attributes":{"field":"links:part_of:discovery-benchmark-tape","target_id":"discovery-benchmark-tape","source_locator":"README: List of Models and Tasks; Leaderboard","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-aa25f55c703d596f37e1","kind":"claim","name":"GlycanML immunogenicity prediction: part of GlycanML","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["glycomics"]},"source_ids":["src-discovery-glycanml-glycanml"],"links":[{"relation":"subject","target_id":"discovery-benchmark-glycanml-immunogenicity-prediction"}],"attributes":{"field":"links:part_of:discovery-benchmark-glycanml","target_id":"discovery-benchmark-glycanml","source_locator":"README: Overview, task list","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-aaccdc5a8a938e70518e","kind":"claim","name":"MFASS v1 (superseded): evaluates task MFASS splice-variant prioritisation","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"subject","target_id":"rewire-mfass-v1"}],"attributes":{"field":"links:evaluates_task:catalog-task-mfass-splice","target_id":"catalog-task-mfass-splice","source_locator":"Pinned benchmarks/mfass/README.md: Correction; v1 archived as superseded evidence","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-b77dc7ed3e4c0e21c842","kind":"claim","name":"ESM-2: variant of ESM-2","description":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities.","status":"source_checked","facets":{"areas":["proteins-complexes"],"method_types":["foundation model"]},"source_ids":["src-discovery-facebookresearch-esm"],"links":[{"relation":"subject","target_id":"catalog-model-esm-2"}],"attributes":{"field":"links:variant_of:discovery-model-esm-2","target_id":"discovery-model-esm-2","source_locator":"README.md: Pre-trained Models identifies esm2_t6_8M_UR50D","review":{"method":"automated_source_review","date":"2026-09-16","note":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities."}}} {"id":"profile-association-b98e7ae409d7f83b469e","kind":"claim","name":"DNABERT-2: variant of DNABERT-2","description":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities.","status":"source_checked","facets":{"areas":["dna-genomes"],"method_types":["foundation model"]},"source_ids":["src-discovery-magics-lab-dnabert-2"],"links":[{"relation":"subject","target_id":"catalog-model-dnabert-2"}],"attributes":{"field":"links:variant_of:discovery-model-dnabert-2","target_id":"discovery-model-dnabert-2","source_locator":"README.md: Model and Data names the 117M checkpoint","review":{"method":"automated_source_review","date":"2026-09-16","note":"Verified named released configuration or extension within the official project. This is not alias equivalence and does not move existing result identities."}}} {"id":"profile-association-bff7c057c678296de10c","kind":"claim","name":"CAMI genome binning: part of CAMI","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["microbiome"]},"source_ids":["src-discovery-cami"],"links":[{"relation":"subject","target_id":"discovery-benchmark-cami-genome-binning"}],"attributes":{"field":"links:part_of:discovery-benchmark-cami","target_id":"discovery-benchmark-cami","source_locator":"CAMI official home page: Summary / Per category","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-ceb267609f7dffa0b950","kind":"claim","name":"MassSpecGym Spectrum simulation: part of MassSpecGym","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["metabolomics"]},"source_ids":["src-discovery-pluskal-lab-massspecgym"],"links":[{"relation":"subject","target_id":"discovery-benchmark-massspecgym-spectrum-simulation"}],"attributes":{"field":"links:part_of:discovery-benchmark-massspecgym","target_id":"discovery-benchmark-massspecgym","source_locator":"README: opening challenge list","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-d69e83003c6ac3369c10","kind":"claim","name":"TAPE Stability: part of TAPE","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["protein-function"]},"source_ids":["src-discovery-songlab-cal-tape"],"links":[{"relation":"subject","target_id":"discovery-benchmark-tape-stability"}],"attributes":{"field":"links:part_of:discovery-benchmark-tape","target_id":"discovery-benchmark-tape","source_locator":"README: List of Models and Tasks; Leaderboard","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-dcbdd6f96b71ea87cb77","kind":"claim","name":"TAPE Fluorescence: part of TAPE","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["protein-function"]},"source_ids":["src-discovery-songlab-cal-tape"],"links":[{"relation":"subject","target_id":"discovery-benchmark-tape-fluorescence"}],"attributes":{"field":"links:part_of:discovery-benchmark-tape","target_id":"discovery-benchmark-tape","source_locator":"README: List of Models and Tasks; Leaderboard","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-dd1df650aeabc9d220d1","kind":"claim","name":"CAMI taxonomic profiling: part of CAMI","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["microbiome"]},"source_ids":["src-discovery-cami"],"links":[{"relation":"subject","target_id":"discovery-benchmark-cami-taxonomic-profiling"}],"attributes":{"field":"links:part_of:discovery-benchmark-cami","target_id":"discovery-benchmark-cami","source_locator":"CAMI official home page: Summary / Per category","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-df9d68d965eef2930ace","kind":"claim","name":"ProteinGym: evaluates task ProteinGym mutation effects","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["protein-function"]},"source_ids":["src-discovery-oatml-markslab-proteingym"],"links":[{"relation":"subject","target_id":"discovery-benchmark-proteingym"}],"attributes":{"field":"links:evaluates_task:catalog-task-proteingym-effects","target_id":"catalog-task-proteingym-effects","source_locator":"README: Overview; Results","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-e433e3e33937f0e3c53a","kind":"claim","name":"CAMI taxonomic binning: part of CAMI","description":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence.","status":"source_checked","facets":{"areas":["microbiome"]},"source_ids":["src-discovery-cami"],"links":[{"relation":"subject","target_id":"discovery-benchmark-cami-taxonomic-binning"}],"attributes":{"field":"links:part_of:discovery-benchmark-cami","target_id":"discovery-benchmark-cami","source_locator":"CAMI official home page: Summary / Per category","review":{"method":"automated_source_review","date":"2026-09-16","note":"Navigation relationship checked against the named primary-source section; not equivalence of protocol versions or independent evidence."}}} {"id":"profile-association-e63a9e1f6beae6543100","kind":"claim","name":"Pangolin: family Pangolin","description":"Both records point to the same official project; the catalogue record has no independently identified checkpoint. Preserve both IDs and restrict shared explanation to family-level facts.","status":"source_checked","facets":{"areas":["dna-genomes"],"method_types":["specialist"]},"source_ids":["src-discovery-tkzeng-pangolin"],"links":[{"relation":"subject","target_id":"catalog-model-pangolin"}],"attributes":{"field":"links:family:discovery-model-pangolin","target_id":"discovery-model-pangolin","source_locator":"Official README project identity and the existing catalogue source URL","review":{"method":"automated_source_review","date":"2026-09-16","note":"Both records point to the same official project; the catalogue record has no independently identified checkpoint. Preserve both IDs and restrict shared explanation to family-level facts."}}} {"id":"prokbert-2024","kind":"source","name":"ProkBERT family: genomic language models for microbiome applications","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10810988/","version":"PMC10810988.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.3389/fmicb.2023.1331233","publication_status":"peer_reviewed","year":2024,"artifact_sha256":"8610e2a54aa877c8dc565a9cdb6e82099f284c5e0907a52cab18d994ea732436","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10810988/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:36.197Z","legacy_paper":{"id":"prokbert-2024","title":"ProkBERT family: genomic language models for microbiome applications","year":2024,"publication_status":"peer_reviewed","version":"PMC10810988.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10810988/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Frontiers in Microbiology; PMC ID: PMC10810988.","doi":"10.3389/fmicb.2023.1331233"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"protein-binding-sites-2023","kind":"source","name":"Learning the protein language of proteome-wide protein-protein binding sites via explainable ensemble deep learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9849350/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1038/s42003-023-04462-5","publication_status":"peer_reviewed","year":2023,"artifact_sha256":"491711aa7186e74bf33f6d601c4ea6a8f565e770938fe1f91a0cd347b8f06f3f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC9849350/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"protein-binding-sites-2023","title":"Learning the protein language of proteome-wide protein-protein binding sites via explainable ensemble deep learning","year":2023,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC9849350/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Communications Biology; PMC ID: PMC9849350. Downstream binding-site classifier; not a native ProtT5 prediction head.","doi":"10.1038/s42003-023-04462-5"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"proteingym-2023","kind":"source","name":"ProteinGym: Large-Scale Benchmarks for Protein Design and Fitness Prediction","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10723403/","version":"PMC10723403.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2023.12.07.570727","publication_status":"preprint","year":2023,"artifact_sha256":"4519641f13271bdd09b166e7d93232f22542489bc52a25b5a1628c3df8badce1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10723403/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.517323+00:00","legacy_paper":{"id":"proteingym-2023","title":"ProteinGym: Large-Scale Benchmarks for Protein Design and Fitness Prediction","year":2023,"publication_status":"preprint","version":"PMC10723403.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10723403/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC10723403.","doi":"10.1101/2023.12.07.570727"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"pst-2025","kind":"source","name":"Endowing protein language models with structural knowledge","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12603367/","version":"PMC12603367.1","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1093/bioinformatics/btaf582","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"c21ad593de589a7188ca86a8b7ce617e301d48dd939ecc7da03efb22cbe8d7a3","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12603367/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.527633+00:00","legacy_paper":{"id":"pst-2025","title":"Endowing protein language models with structural knowledge","year":2025,"publication_status":"peer_reviewed","version":"PMC12603367.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12603367/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Bioinformatics; PMC ID: PMC12603367.","doi":"10.1093/bioinformatics/btaf582"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"quadruplex-llm-benchmark-2025","kind":"source","name":"Benchmarking DNA large language models on quadruplexes","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["dna-genomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11953744/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1016/j.csbj.2025.03.007","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"c3d7c6d068d3c11a9c8255a932197ece3804d94e8b2d4f858bea373a1b6eb32f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11953744/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:35.379Z","legacy_paper":{"id":"quadruplex-llm-benchmark-2025","title":"Benchmarking DNA large language models on quadruplexes","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11953744/","primary_domain":"dna-genomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Computational and Structural Biotechnology Journal; PMC ID: PMC11953744.","doi":"10.1016/j.csbj.2025.03.007"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"r3design-2025","kind":"source","name":"R3Design: deep tertiary structure-based RNA sequence design and beyond","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11685104/","version":"PMC archival version PMC11685104.1","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/bib/bbae682","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"b49c9ad846e46b11b240aace8e6bfaf953b842df166b69aee4843c02e9349779","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11685104/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"r3design-2025","title":"R3Design: deep tertiary structure-based RNA sequence design and beyond","year":2025,"publication_status":"peer_reviewed","version":"PMC archival version PMC11685104.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11685104/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Briefings in Bioinformatics; PMC ID: PMC11685104. Architecture and task differ from RNA language-model encoding benchmarks.","doi":"10.1093/bib/bbae682"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-00201f65c32f6d","kind":"dataset","name":"α-synuclein Ligand 47 MD ensemble","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["ensemble-idp-docking-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-034c60a2dabc73","kind":"dataset","name":"genome-wide TF binding sites","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["transbind-2026"],"links":[],"attributes":{"version":null,"split":"test","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-065b9fcc8da573","kind":"dataset","name":"CASF-2016 blind docked poses","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["nmdn-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-07d355c146be1f","kind":"dataset","name":"paper PPI test set","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["gsmformer-ppi-2026"],"links":[],"attributes":{"version":null,"split":"test set","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-0aab382ca2c063","kind":"dataset","name":"CLA-IND0.6","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["clathrin-plm-2025"],"links":[],"attributes":{"version":null,"split":"independent test","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-0b54f42a987b1d","kind":"dataset","name":"testing viral metagenome dataset","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["detire-viral-metagenomes-2023"],"links":[],"attributes":{"version":null,"split":"test","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-0bba1c9a7ae410","kind":"dataset","name":"genomic benchmark categories","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[],"attributes":{"version":null,"split":"paper benchmark summary","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-0c3ac7efe99c37","kind":"dataset","name":"GenomeOcean natural/artificial sequence test","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["genomeocean-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-0edd8f724db696","kind":"dataset","name":"antibody peptide-mapping training dataset","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[],"attributes":{"version":null,"split":"fivefold stratified CV","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-158b121281b650","kind":"dataset","name":"Yoruban LCL dsQTLs","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["arsenal-regulatory-dna-2026"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-16d01b5ef88e84","kind":"dataset","name":"Fingerprint-scoring benchmark","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["fingerprint-scoring-2022"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-17132fbabd7683","kind":"dataset","name":"CASF-2016","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["ligand-affinity-meta-model-2024"],"links":[],"attributes":{"version":null,"split":"core benchmark","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-1744719eef145b","kind":"dataset","name":"human ultra-long mRNAs","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["mrnabert-2025"],"links":[],"attributes":{"version":null,"split":"paper evaluation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-18ebde58579c2b","kind":"dataset","name":"DEBFold TestSetβ","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["debfold-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-1b4f6ea24c0587","kind":"dataset","name":"LAMBDA genome-wide prophage test","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["lambda-prophage-2026"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-1c4c71078ffe01","kind":"dataset","name":"MetaHIT","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["metagenomic-pathogens-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-1c7f8ebb1968d9","kind":"dataset","name":"T18","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["rlsite-rna-binding-2025"],"links":[],"attributes":{"version":null,"split":"paper evaluation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-235520c84b737f","kind":"dataset","name":"SARS-CoV-2 Mpro ligands","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["mpro-pose-affinity-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-236eaa4e55147f","kind":"dataset","name":"liver editing sites","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["adar-gpt-editing-2026"],"links":[],"attributes":{"version":null,"split":"15% validation set","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-2ad2fad5e1cd0a","kind":"dataset","name":"hESC cell-type-specific GRN","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scregnet-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-2e87449871ca47","kind":"dataset","name":"mRNA-RBP pairs","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["mrna-protein-diversity-2026"],"links":[],"attributes":{"version":null,"split":"RBP-aware test set","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-2eaa2a051d45ee","kind":"dataset","name":"variant-effects benchmark","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["structure-informed-plm-2025"],"links":[],"attributes":{"version":null,"split":"paper evaluation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-327cfcdae0c937","kind":"dataset","name":"PDBbind core v2016","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["deelig-2021"],"links":[],"attributes":{"version":"v2016","split":null,"missing_metadata":{"split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-32ccef507a1dd7","kind":"dataset","name":"ncRNA interaction pairs","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["cupid-rna-interactions-2026"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-33a41fe5fc66cf","kind":"dataset","name":"PBMCs-BS","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scalr-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-38151fa548e291","kind":"dataset","name":"HumanPPI","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["mulan-2025"],"links":[],"attributes":{"version":null,"split":"paper evaluation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-3a3e3880a3fed0","kind":"dataset","name":"mRNABench MRL-MPRA","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["mrnabench-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-43f24c4dfb7351","kind":"dataset","name":"human and viral proteins","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["viral-immune-mimicry-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-450c1af18cc623","kind":"dataset","name":"Simulated viral metagenome","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["lazypipe-2020"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-463197d6a98b99","kind":"dataset","name":"Human 5mC","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["dna-foundation-models-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-477a9082515406","kind":"dataset","name":"Human thymus scRNA-seq","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["mouse-geneformer-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-488d5de6bb9c1b","kind":"dataset","name":"M.S. single-cell dataset","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["single-cell-peft-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-48def1da574597","kind":"dataset","name":"E. coli sigma70 promoter dataset","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["prokbert-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-4b6c13924d4256","kind":"dataset","name":"PoseBusters","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["molas-2026"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-50f0bdb7cf9ca4","kind":"dataset","name":"HIV neutralization","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["deepinteraware-2025"],"links":[],"attributes":{"version":null,"split":"antibody-unseen","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-5197cca532f89d","kind":"dataset","name":"FUJISAN test sub-dataset","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["fujisan-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-52f00ccaabf0d9","kind":"dataset","name":"mRNA half-life","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["mrna-lm-2025"],"links":[],"attributes":{"version":null,"split":"test set across CV splits","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-54b9bc432928d6","kind":"dataset","name":"flu-vaccine sequences","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["codonbert-vaccines-2024"],"links":[],"attributes":{"version":null,"split":"paper evaluation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-55f200c9481409","kind":"dataset","name":"poly(A) Gene-Gene","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["polya-glm-2025"],"links":[],"attributes":{"version":null,"split":"5-fold cross-validation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-561834dfa1682c","kind":"dataset","name":"ICCTax Complete dataset","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["icctax-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-571ce000cd74b9","kind":"dataset","name":"AIDA v2 PBMC cohort","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["single-cell-aging-probes-2026"],"links":[],"attributes":{"version":"622 donors","split":null,"missing_metadata":{"split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-59def895fbdbb4","kind":"dataset","name":"non-immune cells","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["gremln-2026"],"links":[],"attributes":{"version":null,"split":"zero-shot","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-5afaefb87c8a94","kind":"dataset","name":"PLINDER-L95","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["boltz-stereochemistry-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-6212e779949708","kind":"dataset","name":"Liu training dataset","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["dnabert2-enhancer-2025"],"links":[],"attributes":{"version":null,"split":"5-fold cross-validation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-6a44f5946cd7ab","kind":"dataset","name":"LiPP lipid–protein complexes","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["lipp-2026"],"links":[],"attributes":{"version":"331 complexes","split":null,"missing_metadata":{"split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-6e0c28dfde7337","kind":"dataset","name":"CATH superfamily benchmark","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["cathe2-2025"],"links":[],"attributes":{"version":null,"split":"paper evaluation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-701d910b02d25c","kind":"dataset","name":"SJC","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["clape-smb-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-71614d99b3099f","kind":"dataset","name":"Rfam","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["r3design-2025"],"links":[],"attributes":{"version":null,"split":"external","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-72e837e5b97041","kind":"dataset","name":"HumanGut-all strain-level query","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["cammiq-2022"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-739aee3cf8d6f1","kind":"dataset","name":"DNA-129_Test","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["megsite-2025"],"links":[],"attributes":{"version":null,"split":"independent test","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-768a7ff5bac414","kind":"dataset","name":"Zymo LOG 10%","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["lemur-magnet-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-7bf2cf7d2b2d01","kind":"dataset","name":"stimulated immune PBMC","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["insilico-perturbation-auprc-2025"],"links":[],"attributes":{"version":null,"split":"CD14+Mono","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-7cec655cd742f3","kind":"dataset","name":"ProteinGym substitution DMS: stability assays","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["proteingym-2023"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-7fc59ce4c0ceaa","kind":"dataset","name":"MosA1 reference → WholeBrainA query","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scatac-llmda-2026"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-8317793f18b026","kind":"dataset","name":"PDB RNA set","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["bpfold-2025"],"links":[],"attributes":{"version":"116 RNAs","split":null,"missing_metadata":{"split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-87e91d9d6e6f4b","kind":"dataset","name":"L1000","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["cell2sentence-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-8cad416ddc80dc","kind":"dataset","name":"Real mock community MAGs","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["kmetashot-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-8e9488896becd4","kind":"dataset","name":"GUE H-CPD","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["eden-genomic-classification-2026"],"links":[],"attributes":{"version":null,"split":"paper evaluation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-8f123f006964ad","kind":"dataset","name":"hPancreas","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scelmo-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-9135087a16af1c","kind":"dataset","name":"Bernett PPI dataset","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["esm2-amp-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-99afd0c86b2954","kind":"dataset","name":"MirTarRAW","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["rnaret-2026"],"links":[],"attributes":{"version":null,"split":"held-out test","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-9c186c8f4ed3f4","kind":"dataset","name":"ProteinGym substitutions","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[],"attributes":{"version":null,"split":"aggregate across assays","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-9e9d18bc5bfb8b","kind":"dataset","name":"KEx","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-a03b8e9efde37b","kind":"dataset","name":"human enhancer dataset","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["enhancer-position-encoding-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-a1da4a37eb46a5","kind":"dataset","name":"Independent E. coli sigma70 test dataset","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["cyaprombert-2022"],"links":[],"attributes":{"version":null,"split":"independent test","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-a28180d33f7a23","kind":"dataset","name":"ClinVar 3-prime UTR variants","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["phylogpn-2025"],"links":[],"attributes":{"version":null,"split":"paper evaluation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-a8610f2b80cdf0","kind":"dataset","name":"enhancer independent comparison","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["hi-enhancer-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-abdfba8cce7486","kind":"dataset","name":"bpRNA-new","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["ernie-rna-2025"],"links":[],"attributes":{"version":null,"split":"cross-family test","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-b1af840b76b351","kind":"dataset","name":"CoBRA compound-binding test set","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["cobra-rna-binding-2026"],"links":[],"attributes":{"version":null,"split":"test set","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-b462aa24561fba","kind":"dataset","name":"CAMI II Toy human gastrooral sample19-new","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["nabas-plus-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-b6ce37ba678d39","kind":"dataset","name":"DART-Eval cCREs versus matched shuffled controls","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["dart-eval-regulatory-2024"],"links":[],"attributes":{"version":null,"split":"paper evaluation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-b6d0ebaca196a6","kind":"dataset","name":"Andropogoneae genome-wide conservation","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["plantcad2-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-b7204b005bd476","kind":"dataset","name":"ImmuneBuilder antibody test set","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["ibex-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-b91c871eb7740a","kind":"dataset","name":"vaccine candidate validation","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["vaxign-esm-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-bc127dc9c441fe","kind":"dataset","name":"DNA barcodes of unseen species","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["barcodebert-2026"],"links":[],"attributes":{"version":null,"split":"1-NN probe","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-bd3d8e7d6cd196","kind":"dataset","name":"human RNA 2OMe sites","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["2ome-lm-2025"],"links":[],"attributes":{"version":null,"split":"5-fold cross-validation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-bd3f98e2eeb5d3","kind":"dataset","name":"CRC microbiome cohort","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["mdl4microbiome-2022"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-bd9255afb783d6","kind":"dataset","name":"ProteinShake VEP datasets","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["pst-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-beb4f5da29da0a","kind":"dataset","name":"CAMI II Sample_0 10,000-read subsample","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["ncd-metagenomics-2026"],"links":[],"attributes":{"version":"10,000 reads","split":null,"missing_metadata":{"split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-becc215358afd0","kind":"dataset","name":"PRIME mutated RBD","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["prime-2026"],"links":[],"attributes":{"version":null,"split":"position-stratified","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-cd51026cdb6a7a","kind":"dataset","name":"20 medium/high-complexity viral simulations","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["viral-contig-simulation-2021"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-d28955d5872903","kind":"dataset","name":"AMP","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["pc-mer-2024"],"links":[],"attributes":{"version":null,"split":"genus-level","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-d356eac961cb69","kind":"dataset","name":"HPA-FoV","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["cell-dino-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-d9fdd8dc7a0184","kind":"dataset","name":"immune tissue","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["single-cell-residual-geometry-2026"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-dba1707164d296","kind":"dataset","name":"TRX","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["spin-protein-function-2026"],"links":[],"attributes":{"version":null,"split":"test set","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-e0f34dcaa1ba3b","kind":"dataset","name":"23 independent prokaryotic promoter test sets","description":"","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["ipromp-2025"],"links":[],"attributes":{"version":"23 test sets","split":"independent test","missing_metadata":{"accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-e45a5a140888ee","kind":"dataset","name":"Boltz-1 structure test set","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["boltz1-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-eaa2965545c87b","kind":"dataset","name":"Aorta single-cell dataset","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["genept-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-ebc3f5fda43972","kind":"dataset","name":"extremely long-sequence species classification","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["birna-bert-2025"],"links":[],"attributes":{"version":null,"split":"paper evaluation","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-ee26acbd6e8cf7","kind":"dataset","name":"Antibody–antigen GEP test set","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["antibody-flexibility-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-f08b1a60aebeeb","kind":"dataset","name":"Dset_448","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["protein-binding-sites-2023"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-f0bf60a62ad7c3","kind":"dataset","name":"Genomic Benchmarks Mouse Enhancers","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["enbed-2024"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-f18fcc23dfa798","kind":"dataset","name":"CASF-2016","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["akscore-2020"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-f2e729f333a333","kind":"dataset","name":"RNA8F","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["tu-fold-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-f6922a9744ba27","kind":"dataset","name":"gene fusion breakpoint DNA sequences","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[],"attributes":{"version":null,"split":"full test set","missing_metadata":{"version":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-f7210686a78474","kind":"dataset","name":"scXDR transfer scenario 2","description":"","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scxdr-2026"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-dataset-fcb5752d916d5d","kind":"dataset","name":"DNALongBench ETGP","description":"","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["dnalongbench-2025"],"links":[],"attributes":{"version":null,"split":null,"missing_metadata":{"version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract","accession":"not_reported_in_legacy_extract"}}} {"id":"reported-model-0068c3eff1bf7b","kind":"model","name":"TOPBP (Complex)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["deelig-2021"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"TOPBP (Complex)","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"TOPBP (Complex) is the method recorded for Protein–ligand binding affinity prediction. This page preserves the configuration reported by DEELIG: A Deep Learning Approach to Predict Protein-Ligand Binding Affinity.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Source table compiles a previously published comparator; protocol equivalence is not established.","source_ids":["deelig-2021"],"source_locator":"Table 2, TOPBP (Complex) row, PDBbind v2016 column"}],"facts":[{"label":"Recorded dataset","value":"PDBbind core v2016","source_ids":["deelig-2021"],"source_locator":"Table 2, TOPBP (Complex) row, PDBbind v2016 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["deelig-2021"],"source_locator":"Table 2, TOPBP (Complex) row, PDBbind v2016 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, TOPBP (Complex) row, PDBbind v2016 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-028e4bb9baa074","kind":"model","name":"Single best solver","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["molas-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Single best solver","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Single best solver is the method recorded for Physically valid protein–ligand pose selection. This page preserves the configuration reported by Molecular embedding-based algorithm selection in protein-ligand docking.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Single best solver baseline under the same averaged five-fold selection test.","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"}],"facts":[{"label":"Recorded dataset","value":"PoseBusters","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, PoseBusters / Mixed / AutoDock row, SBS success column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-03080a5289c07e","kind":"model","name":"Prompt","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["ipromp-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Prompt","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Prompt is the method recorded for Multi-species prokaryotic promoter detection. This page preserves the configuration reported by iPro-MP: a BERT-based model to predict multiple prokaryotic promoters.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Average over the same independent testing sets.","source_ids":["ipromp-2025"],"source_locator":"Table 2, Prompt row, AUC column"}],"facts":[{"label":"Recorded dataset","value":"23 independent prokaryotic promoter test sets","source_ids":["ipromp-2025"],"source_locator":"Table 2, Prompt row, AUC column"},{"label":"Recorded split","value":"independent test","source_ids":["ipromp-2025"],"source_locator":"Table 2, Prompt row, AUC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["ipromp-2025"],"source_locator":"Table 2, Prompt row, AUC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Prompt row, AUC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-035a3ab36a3a6a","kind":"model","name":"structure-informed pLM","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["structure-informed-plm-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"structure-informed pLM","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"structure-informed pLM is the method recorded for protein variant-effect classification. This page preserves the configuration reported by Structure-Informed Protein Language Models are Robust Predictors for Variant Effects.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: combined amino-acid, secondary structure, solvent accessibility and contact-map scoring","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"variant-effects benchmark","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"},{"label":"Recorded split","value":"paper evaluation","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-05103f72325fe5","kind":"model","name":"BarcodeBERT (4–4-4)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["barcodebert-2026"],"links":[],"attributes":{"entity_level":"method","version":"4–4–4","reported_name":"BarcodeBERT (4–4-4)","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"BarcodeBERT learns DNA-barcode representations for biodiversity analysis. This record is the paper’s four-layer, four-head, 4-mer configuration.","sections":[{"title":"How it works","body":"Non-overlapping DNA 4-mers enter four transformer layers. Masked-token pretraining uses barcode sequences, with random offsets to reduce tokenisation sensitivity. Average pooling produces a barcode embedding. The linked evaluation compares embeddings by cosine similarity to assign a genus from the nearest reference.","source_ids":["barcodebert-2026"],"source_locator":"Sections 3.1–3.2, Figure 2, Section 4.1.3 and Table 1; PMC13008329 fullTextXML"}],"facts":[{"label":"Input preparation","value":"Pad or truncate to 660 nucleotides","source_ids":["barcodebert-2026"],"source_locator":"Sections 3.1–3.2, Figure 2, Section 4.1.3 and Table 1; PMC13008329 fullTextXML"},{"label":"Training source","value":"Canadian invertebrate DNA barcodes","source_ids":["barcodebert-2026"],"source_locator":"Sections 3.1–3.2, Figure 2, Section 4.1.3 and Table 1; PMC13008329 fullTextXML"},{"label":"Configuration in this record","value":"4–4–4","source_ids":["barcodebert-2026"],"source_locator":"Sections 3.1–3.2, Figure 2, Section 4.1.3 and Table 1; PMC13008329 fullTextXML"}],"strengths":[{"text":"Barcode-specific pretraining supports genus assignment for species excluded from the reference training partition.","source_ids":["barcodebert-2026"],"source_locator":"Sections 3.1–3.2, Figure 2, Section 4.1.3 and Table 1; PMC13008329 fullTextXML"}],"limitations":[{"text":"The unseen-species test retains known genera. It is not a demonstration of recognising entirely new genera; BLAST remains a meaningful comparator.","source_ids":["barcodebert-2026"],"source_locator":"Sections 3.1–3.2, Figure 2, Section 4.1.3 and Table 1; PMC13008329 fullTextXML"}],"diagram":{"title":"Conceptual procedure","steps":["COI barcode","Non-overlapping 4-mers","Four transformer layers","Average embedding","Genus by 1-nearest neighbour"],"caption":"Schematic of the documented input, computation and output; not an executable configuration.","source_ids":["barcodebert-2026"],"source_locator":"Sections 3.1–3.2, Figure 2, Section 4.1.3 and Table 1; PMC13008329 fullTextXML"},"coverage":"reviewed","gaps":["Exact released weight hash is not linked to the table row.","The paper table’s 78.5% is genus-level 1-NN accuracy, not seen-species accuracy."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary project documentation or paper inspected for the explanatory claims and cited locations. Reviewed coverage concerns this narrative, not complete metadata, independent reproduction or a performance ranking."}}}} {"id":"reported-model-06816ce9073144","kind":"model","name":"Geneformer","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["cell2sentence-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Geneformer","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Geneformer is the method recorded for Combinatorial cell-label classification. This page preserves the configuration reported by Cell2Sentence: Teaching Large Language Models the Language of Biology.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Partial-credit labels including cell type, perturbation, and dose.","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / Geneformer row, L1000 Acc column"}],"facts":[{"label":"Recorded dataset","value":"L1000","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / Geneformer row, L1000 Acc column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / Geneformer row, L1000 Acc column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, Partial label / Geneformer row, L1000 Acc column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-0829aff5471d4b","kind":"model","name":"Stacking-Auto","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["hi-enhancer-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Stacking-Auto","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Stacking-Auto is the method recorded for enhancer prediction. This page preserves the configuration reported by Hi-Enhancer: a two-stage framework for prediction and localization of enhancers based on Blending-KAN and Stacking-Auto models.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Two-stage Hi-Enhancer system; paper Table 2 method comparison","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"}],"facts":[{"label":"Recorded dataset","value":"enhancer independent comparison","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Ours (Stacking-Auto) row, Accuracy column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-0d147487bf97be","kind":"model","name":"Promotech","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["prokbert-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Promotech","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Promotech is the method recorded for E. coli sigma70 promoter prediction. This page preserves the configuration reported by ProkBERT family: genomic language models for microbiome applications.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Promoter versus non-promoter classification.","source_ids":["prokbert-2024"],"source_locator":"Table 3, Promotech row, Accuracy column"}],"facts":[{"label":"Recorded dataset","value":"E. coli sigma70 promoter dataset","source_ids":["prokbert-2024"],"source_locator":"Table 3, Promotech row, Accuracy column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["prokbert-2024"],"source_locator":"Table 3, Promotech row, Accuracy column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, Promotech row, Accuracy column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-0eb4b0535b58e3","kind":"model","name":"MolAS","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["molas-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"MolAS","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"MolAS is the method recorded for Physically valid protein–ligand pose selection. This page preserves the configuration reported by Molecular embedding-based algorithm selection in protein-ligand docking.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Averaged five-fold algorithm-selection performance on PoseBusters; joint RMSD and validity criterion.","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column"}],"facts":[{"label":"Recorded dataset","value":"PoseBusters","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-10d85f2a035720","kind":"model","name":"Human-Geneformer","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["mouse-geneformer-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Human-Geneformer","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Human-Geneformer is the method recorded for Human thymus cell-type classification. This page preserves the configuration reported by Mouse-Geneformer: A deep learning model for mouse single-cell transcriptome and its cross-species utility.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Native human model; zero-shot setting.","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"}],"facts":[{"label":"Recorded dataset","value":"Human thymus scRNA-seq","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-13bd2a6c2d8178","kind":"model","name":"mRNABERT","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["mrnabert-2025"],"links":[],"attributes":{"entity_level":"method","version":"3066-nt input","reported_name":"mRNABERT","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"mRNABERT is the method recorded for translation-efficiency prediction. This page preserves the configuration reported by mRNABERT: advancing mRNA sequence design with a universal language model and comprehensive dataset.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: human translation-efficiency regression at 3066-nt input","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"}],"facts":[{"label":"Recorded dataset","value":"human ultra-long mRNAs","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"},{"label":"Recorded split","value":"paper evaluation","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"},{"label":"Recorded configuration","value":"3066-nt input","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, mRNABERT (3066) row, Human R-squared column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-148b613975b6eb","kind":"model","name":"Best frozen single-cell foundation model","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["single-cell-aging-probes-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Best frozen single-cell foundation model","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Best frozen single-cell foundation model is the method recorded for Donor-aware age-class prediction. This page preserves the configuration reported by Inflammation-linked aging signals in frozen single-cell foundation models: donor-aware detection and robustness testing.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Same donor-aware splits and logistic-regression probe as expression PCA; text names Geneformer as best model on AIDA v2.","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column"}],"facts":[{"label":"Recorded dataset","value":"AIDA v2 PBMC cohort","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, AIDA v2 row, scFM BA ± SD column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-1587ab674d30a2","kind":"model","name":"AutoDock Vina holo","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["ensemble-idp-docking-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"AutoDock Vina holo","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"AutoDock Vina holo is the method recorded for Intrinsically disordered protein ensemble docking. This page preserves the configuration reported by Ensemble docking for intrinsically disordered proteins.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble.","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column"}],"facts":[{"label":"Recorded dataset","value":"α-synuclein Ligand 47 MD ensemble","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Ligand 47 row, AutoDock Vina Holo Docking column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-177f32ce8189a0","kind":"model","name":"Gene-expression PCA","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["single-cell-aging-probes-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Gene-expression PCA","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Gene-expression PCA is the method recorded for Donor-aware age-class prediction. This page preserves the configuration reported by Inflammation-linked aging signals in frozen single-cell foundation models: donor-aware detection and robustness testing.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Fifty-component gene-expression PCA with the same donor-aware probe splits.","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, Gene-expr BA column"}],"facts":[{"label":"Recorded dataset","value":"AIDA v2 PBMC cohort","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, Gene-expr BA column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, Gene-expr BA column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, AIDA v2 row, Gene-expr BA column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-1a67087ac262c5","kind":"model","name":"Nucleotide Transformer + NN (middle)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"Nucleotide Transformer + NN (middle)","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Nucleotide Transformer + NN (middle) is the method recorded for gene fusion breakpoint classification. This page preserves the configuration reported by Benchmarking genomic foundation models for binary classification of gene fusion breakpoints from DNA sequences.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: middle embedding with neural-network classifier","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"}],"facts":[{"label":"Recorded dataset","value":"gene fusion breakpoint DNA sequences","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"},{"label":"Recorded split","value":"full test set","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, NT / NN (middle) row, ROC AUC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-1b5fa066945d3d","kind":"model","name":"R3Design","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["r3design-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"R3Design","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"R3Design is the method recorded for RNA sequence design. This page preserves the configuration reported by R3Design: deep tertiary structure-based RNA sequence design and beyond.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Tertiary-structure-conditioned RNA sequence design; external Rfam assessment","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"}],"facts":[{"label":"Recorded dataset","value":"Rfam","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"},{"label":"Recorded split","value":"external","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, R3Design row, Recovery (%) > Rfam column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-1be5c4b7c52a41","kind":"model","name":"scVI + scANVI","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scalr-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scVI + scANVI","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scVI + scANVI is the method recorded for PBMC cell-type classification. This page preserves the configuration reported by scaLR: a low-resource deep neural network-based platform for single cell analysis and biomarker discovery.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: All features and samples from PBMCs-BS; comparison pipeline combines scVI and scANVI.","source_ids":["scalr-2025"],"source_locator":"Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"}],"facts":[{"label":"Recorded dataset","value":"PBMCs-BS","source_ids":["scalr-2025"],"source_locator":"Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["scalr-2025"],"source_locator":"Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-1d2aa9880a1c77","kind":"model","name":"ADAR-GPT continual","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["adar-gpt-editing-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ADAR-GPT continual","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ADAR-GPT continual is the method recorded for A-to-I RNA editing site prediction. This page preserves the configuration reported by ADAR-GPT: A continually fine-tuned language model for predicting A-to-I RNA editing sites.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Curriculum plus 15% fine-tuning; 201-nt sequence windows; decision threshold 0.5","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"}],"facts":[{"label":"Recorded dataset","value":"liver editing sites","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"},{"label":"Recorded split","value":"15% validation set","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Adar-GPT (continual) row, F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-1e51ccbfd2de61","kind":"model","name":"Vina","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["nmdn-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Vina","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Vina is the method recorded for Protein–ligand virtual screening. This page preserves the configuration reported by Normalized Protein–Ligand Distance Likelihood Score for End-to-End Blind Docking and Virtual Screening.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Vina scoring on the same DiffDock-NMDN blind docked poses; not ligand-pose RMSD.","source_ids":["nmdn-2025"],"source_locator":"Table 2, Vina scoring row, success rate (%) column"}],"facts":[{"label":"Recorded dataset","value":"CASF-2016 blind docked poses","source_ids":["nmdn-2025"],"source_locator":"Table 2, Vina scoring row, success rate (%) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["nmdn-2025"],"source_locator":"Table 2, Vina scoring row, success rate (%) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Vina scoring row, success rate (%) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-23cb15b93c00ff","kind":"model","name":"iPro70-FMWin","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["cyaprombert-2022"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"iPro70-FMWin","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"iPro70-FMWin is the method recorded for E. coli sigma70 promoter prediction. This page preserves the configuration reported by TSSNote-CyaPromBERT: Development of an integrated platform for highly accurate promoter prediction and visualization of Synechococcus sp. and Synechocystis sp. through a state-of-the-art natural language processing model BERT.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Compared on the same independent test dataset; 110 promoters and 108 non-promoters.","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, iPro70-FMWin row, F1 score Promoter column"}],"facts":[{"label":"Recorded dataset","value":"Independent E. coli sigma70 test dataset","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, iPro70-FMWin row, F1 score Promoter column"},{"label":"Recorded split","value":"independent test","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, iPro70-FMWin row, F1 score Promoter column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, iPro70-FMWin row, F1 score Promoter column"}],"coverage":"limited","gaps":["The retained evidence location (TABLE 3, iPro70-FMWin row, F1 score Promoter column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-27dca28a87cf3c","kind":"model","name":"VIBRANT","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["viral-contig-simulation-2021"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"VIBRANT","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"VIBRANT is the method recorded for Simulated prophage-contig detection. This page preserves the configuration reported by Simulation study and comparative evaluation of viral contiguous sequence identification tools.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Average across twenty medium- and high-complexity simulated communities.","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column"}],"facts":[{"label":"Recorded dataset","value":"20 medium/high-complexity viral simulations","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, Vibrant row, Prophage F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-28413ae1766316","kind":"model","name":"DNABERT-2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["dart-eval-regulatory-2024"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"DNABERT-2","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DNABERT-2 is the method recorded for regulatory element identification. This page preserves the configuration reported by DART-Eval: A Comprehensive DNA Language Model Evaluation Benchmark on Regulatory DNA.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: zero-shot likelihood ranking: higher likelihood for cCRE than matched control","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"}],"facts":[{"label":"Recorded dataset","value":"DART-Eval cCREs versus matched shuffled controls","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"},{"label":"Recorded split","value":"paper evaluation","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-2894d253c5e8a8","kind":"model","name":"CUPID Data-aug-Avg","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["cupid-rna-interactions-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"CUPID Data-aug-Avg","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"CUPID Data-aug-Avg is the method recorded for non-coding RNA pairwise interaction prediction. This page preserves the configuration reported by Computational understanding of non-coding RNA pairwise interactions.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Data augmentation with average pooling for molecule-level ncRNA embeddings","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"ncRNA interaction pairs","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, CUPID > Data-aug-Avg row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-2ae5fb0c147618","kind":"model","name":"Ibex","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["ibex-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Ibex","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Ibex is the method recorded for Antibody loop structure prediction. This page preserves the configuration reported by Conformation-aware structure prediction of antigen-recognizing immune proteins.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Backbone RMSD after framework alignment; average over antibody test structures.","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column"}],"facts":[{"label":"Recorded dataset","value":"ImmuneBuilder antibody test set","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, Antibodies / Ibex row, CDR H3 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-2cb8118b4c77c0","kind":"model","name":"DNABERT-2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["eden-genomic-classification-2026"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"DNABERT-2","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DNABERT-2 is the method recorded for human core-promoter classification. This page preserves the configuration reported by EDEN: multiscale expected density of nucleotide encoding for enhanced DNA sequence classification with hybrid deep learning.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: DNABERT-2 comparator in consolidated H-CPD table; rerun provenance not explicit","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"}],"facts":[{"label":"Recorded dataset","value":"GUE H-CPD","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"},{"label":"Recorded split","value":"paper evaluation","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, DNABERT-2 row, H-CPD (MCC) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-2df975e60d16d1","kind":"model","name":"GenomeOcean","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["genomeocean-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"GenomeOcean","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"GenomeOcean is the method recorded for Natural vs artificial microbial genome sequence. This page preserves the configuration reported by GenomeOcean: An Efficient Genome Foundation Model Trained on Large-Scale Metagenomic Assemblies.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Source reports natural-versus-artificial sequence classification.","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column"}],"facts":[{"label":"Recorded dataset","value":"GenomeOcean natural/artificial sequence test","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, GenomeOcean row, F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-32a19f43a4c254","kind":"model","name":"CAMMiQ","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["cammiq-2022"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"CAMMiQ","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"CAMMiQ is the method recorded for Strain-level abundance quantification. This page preserves the configuration reported by Strain level microbial detection and quantification with applications to single cell metagenomics.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Strain-level quantification on the HumanGut-all synthetic query.","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column"}],"facts":[{"label":"Recorded dataset","value":"HumanGut-all strain-level query","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-399b1ce87a3f6d","kind":"model","name":"scGPT","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["genept-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scGPT","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scGPT is the method recorded for Cell-type structure in frozen embeddings. This page preserves the configuration reported by GenePT: A Simple But Effective Foundation Model for Genes and Cells Built From ChatGPT.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: k-means on pretrained cell embeddings; agreement with original cell-type labels.","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, scGPT ARI column"}],"facts":[{"label":"Recorded dataset","value":"Aorta single-cell dataset","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, scGPT ARI column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, scGPT ARI column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Aorta / Cell type row, scGPT ARI column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-3af86cb274f658","kind":"model","name":"NT-v2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["dna-foundation-models-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"NT-v2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"NT-v2 is the method recorded for Human 5mC detection. This page preserves the configuration reported by Benchmarking DNA foundation models for genomic and genetic tasks.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Binary epigenetic-modification classification as reported in the paper.","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, NT-v2 column"}],"facts":[{"label":"Recorded dataset","value":"Human 5mC","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, NT-v2 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, NT-v2 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, Human 5mC row, NT-v2 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-3bdb3093e8d531","kind":"model","name":"scGPT","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["single-cell-peft-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scGPT","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scGPT is the method recorded for Cell-type identification. This page preserves the configuration reported by Parameter-Efficient Fine-Tuning Enhances Adaptation of Single Cell Large Language Model for Cell Type Identification.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Native scLLM cell-type identification as reported in Table 2.","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column"}],"facts":[{"label":"Recorded dataset","value":"M.S. single-cell dataset","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, M.S. / scGPT row, F1-Score column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-3c196326586fa7","kind":"model","name":"PMF + ECFP + PF (LightGBM)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["fingerprint-scoring-2022"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"PMF + ECFP + PF (LightGBM)","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"PMF + ECFP + PF (LightGBM) is the method recorded for Protein–ligand binding energy prediction. This page preserves the configuration reported by Machine-Learning- and Knowledge-Based Scoring Functions Incorporating Ligand and Protein Fingerprints.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Binding-energy model using ligand and protein fingerprints with LightGBM.","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column"}],"facts":[{"label":"Recorded dataset","value":"Fingerprint-scoring benchmark","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, PMF + ECFP + PF / LightGBM row, R column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-3e58d0faf88d2e","kind":"model","name":"RiNALMo","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["mrnabench-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"RiNALMo","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"RiNALMo is the method recorded for Mean ribosome load from MPRA. This page preserves the configuration reported by mRNABench: A curated benchmark for mature mRNA property and function prediction.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Linear probe; mean across ten random seeds.","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column"}],"facts":[{"label":"Recorded dataset","value":"mRNABench MRL-MPRA","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, RiNALMo row, MRL MPRA column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-3f850c08d76410","kind":"model","name":"DEBFold","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["debfold-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"DEBFold","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DEBFold is the method recorded for RNA secondary structure. This page preserves the configuration reported by DEBFold: Computational Identification of RNA Secondary Structures for Sequences across Structural Families Using Deep Learning.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Median F1 on the prepared TestSetβ.","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column"}],"facts":[{"label":"Recorded dataset","value":"DEBFold TestSetβ","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, DEBFold row, TestSetβ F1 (%) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-3fd1e9f6c573b2","kind":"model","name":"GTDB-Tk","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["kmetashot-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"GTDB-Tk","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"GTDB-Tk is the method recorded for Mock-community MAG taxonomy classification. This page preserves the configuration reported by kMetaShot: a fast and reliable taxonomy classifier for metagenome-assembled genomes.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Genus classification of the same MAG set.","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus Gtk column"}],"facts":[{"label":"Recorded dataset","value":"Real mock community MAGs","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus Gtk column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus Gtk column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, F1-score % row, Genus Gtk column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-4058c43eb73b90","kind":"model","name":"Boltz-1","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["boltz1-2025"],"links":[],"attributes":{"entity_level":"method","version":"3 recycling rounds; 200 diffusion steps","reported_name":"Boltz-1","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Boltz-1 is the method recorded for Protein–ligand pose prediction. This page preserves the configuration reported by Boltz-1 Democratizing Biomolecular Interaction Modeling.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Highest-confidence pose from five samples; precomputed MSAs up to 4,096 sequences.","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"}],"facts":[{"label":"Recorded dataset","value":"Boltz-1 structure test set","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"},{"label":"Recorded configuration","value":"3 recycling rounds; 200 diffusion steps","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-40ce004270dee4","kind":"model","name":"ESM2 OFS pseudo-perplexity","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"ESM2 OFS pseudo-perplexity","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ESM2 OFS pseudo-perplexity is the method recorded for protein variant fitness prediction. This page preserves the configuration reported by Pseudo-perplexity in One Fell Swoop for Protein Fitness Estimation.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: authors’ zero-shot ESM2 OFS pseudo-perplexity evaluation; aggregate mean across ProteinGym substitution assays","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"}],"facts":[{"label":"Recorded dataset","value":"ProteinGym substitutions","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"},{"label":"Recorded split","value":"aggregate across assays","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"}],"coverage":"limited","gaps":["The retained evidence location (Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-415ee22f46526c","kind":"model","name":"DiffDock","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["mpro-pose-affinity-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"DiffDock","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DiffDock is the method recorded for Ligand potency prediction using generated poses. This page preserves the configuration reported by A Comparative Study of Deep Learning and Classical Modeling Approaches for Protein–Ligand Binding Pose and Affinity Prediction in Coronavirus Main Proteases.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Potency prediction using DiffDock ligand-pose generation plus paper scoring pipeline; not a native DiffDock affinity score.","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, DiffDock row, Pearson’s R column"}],"facts":[{"label":"Recorded dataset","value":"SARS-CoV-2 Mpro ligands","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, DiffDock row, Pearson’s R column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, DiffDock row, Pearson’s R column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, DiffDock row, Pearson’s R column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-41ae49bb40ed8e","kind":"model","name":"ProteinBERT LLM-encoding model","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["mrna-protein-diversity-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ProteinBERT LLM-encoding model","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ProteinBERT LLM-encoding model is the method recorded for mRNA-protein interaction prediction. This page preserves the configuration reported by Generalizable deep-learning-based mRNA-protein interaction prediction strongly depends on protein diversity.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: LLM encoding of protein partner; RBP-aware partition tests generalization to unseen protein diversity","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"}],"facts":[{"label":"Recorded dataset","value":"mRNA-RBP pairs","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"},{"label":"Recorded split","value":"RBP-aware test set","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, RBP-aware test set row, auROC (%) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-43cf51abca83d1","kind":"model","name":"RNA-FM","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["mrnabench-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"RNA-FM","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"RNA-FM is the method recorded for Mean ribosome load from MPRA. This page preserves the configuration reported by mRNABench: A curated benchmark for mature mRNA property and function prediction.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Linear probe; mean across ten random seeds.","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RNA-FM row, MRL MPRA column"}],"facts":[{"label":"Recorded dataset","value":"mRNABench MRL-MPRA","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RNA-FM row, MRL MPRA column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RNA-FM row, MRL MPRA column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, RNA-FM row, MRL MPRA column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-4438513d9cd42c","kind":"model","name":"ProkBERT-mini","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["prokbert-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ProkBERT-mini","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ProkBERT-mini is the method recorded for E. coli sigma70 promoter prediction. This page preserves the configuration reported by ProkBERT family: genomic language models for microbiome applications.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Promoter versus non-promoter classification.","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column"}],"facts":[{"label":"Recorded dataset","value":"E. coli sigma70 promoter dataset","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, ProkBERT-mini row, Accuracy column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-47521865af7b04","kind":"model","name":"Caduceus-Ph","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["dna-foundation-models-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Caduceus-Ph","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Caduceus-Ph is the method recorded for Human 5mC detection. This page preserves the configuration reported by Benchmarking DNA foundation models for genomic and genetic tasks.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Binary epigenetic-modification classification as reported in the paper.","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column"}],"facts":[{"label":"Recorded dataset","value":"Human 5mC","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, Human 5mC row, Caduceus-Ph column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-4921459942b45f","kind":"model","name":"ESM-2 650M embeddings + classifier","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[],"attributes":{"entity_level":"method","version":"esm2_t33_650m_UR50D","reported_name":"ESM-2 650M embeddings + classifier","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ESM-2 650M embeddings + classifier is the method recorded for antibody deamidation-site prediction. This page preserves the configuration reported by The Accurate Prediction of Antibody Deamidations by Combining High-Throughput Automated Peptide Mapping and Protein Language Model-Based Deep Learning.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: global contextual embeddings only","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"}],"facts":[{"label":"Recorded dataset","value":"antibody peptide-mapping training dataset","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"},{"label":"Recorded split","value":"fivefold stratified CV","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"},{"label":"Recorded configuration","value":"esm2_t33_650m_UR50D","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, Global embeddings only row, Accuracy column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-49bc768f46b366","kind":"model","name":"CATHe2 + ProstT5","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["cathe2-2025"],"links":[],"attributes":{"entity_level":"method","version":"full ProstT5","reported_name":"CATHe2 + ProstT5","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"CATHe2 + ProstT5 is the method recorded for CATH superfamily annotation. This page preserves the configuration reported by CATHe2: Enhanced CATH superfamily detection using ProstT5 and structural alphabets.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: amino-acid and structural alphabet embedding classifier","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"}],"facts":[{"label":"Recorded dataset","value":"CATH superfamily benchmark","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"},{"label":"Recorded split","value":"paper evaluation","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"},{"label":"Recorded configuration","value":"full ProstT5","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, ProstT5 full row, F1 score column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-4c73500c39e9d0","kind":"model","name":"ESM2 650M","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["viral-immune-mimicry-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ESM2 650M","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ESM2 650M is the method recorded for human-versus-viral protein classification. This page preserves the configuration reported by Protein Language Models Expose Viral Immune Mimicry.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: ESM2 650M embedding-based human-virus classifier","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"}],"facts":[{"label":"Recorded dataset","value":"human and viral proteins","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, ESM2 650M row, AUC (%) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-4ce8cae0f2eafc","kind":"model","name":"TCINet + HTRS","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["metagenomic-pathogens-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"TCINet + HTRS","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"TCINet + HTRS is the method recorded for pathogen detection. This page preserves the configuration reported by Enhancing pathogen identification through AI-assisted metagenomic sequencing.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Taxonomy-constrained inference network with hierarchical taxonomy representation","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"}],"facts":[{"label":"Recorded dataset","value":"MetaHIT","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-51ed86132346a0","kind":"model","name":"DiffDock-L","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["lipp-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"DiffDock-L","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DiffDock-L is the method recorded for Lipid–protein binding pose. This page preserves the configuration reported by The LiPP Benchmark Set for Modeling Lipid–Protein Complexes: Comparison of Co-Folding and Docking Methods.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Top-scoring pose; all-atom lipid RMSD below 2 Å.","source_ids":["lipp-2026"],"source_locator":"Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"}],"facts":[{"label":"Recorded dataset","value":"LiPP lipid–protein complexes","source_ids":["lipp-2026"],"source_locator":"Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["lipp-2026"],"source_locator":"Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-52b9c685d99290","kind":"model","name":"RLsite","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["rlsite-rna-binding-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"RLsite","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"RLsite is the method recorded for RNA-small-molecule binding-site prediction. This page preserves the configuration reported by RNA language model and graph attention network for RNA and small molecule binding sites prediction.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: RNA language-model plus graph-attention classifier","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"}],"facts":[{"label":"Recorded dataset","value":"T18","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"},{"label":"Recorded split","value":"paper evaluation","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, RLsite row, T18 AUC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-52eee4cc67ca26","kind":"model","name":"RNAfold","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["bpfold-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"RNAfold","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"RNAfold is the method recorded for RNA secondary structure. This page preserves the configuration reported by Deep generalizable prediction of RNA secondary structure via base pair motif energy.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Family-wise evaluation of canonical base-pair predictions.","source_ids":["bpfold-2025"],"source_locator":"Table 2, RNAfold row, PDB F1 column"}],"facts":[{"label":"Recorded dataset","value":"PDB RNA set","source_ids":["bpfold-2025"],"source_locator":"Table 2, RNAfold row, PDB F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["bpfold-2025"],"source_locator":"Table 2, RNAfold row, PDB F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, RNAfold row, PDB F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-53d6515bcc1f39","kind":"model","name":"GREmLN","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["gremln-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"GREmLN","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"GREmLN is the method recorded for cell-type annotation. This page preserves the configuration reported by GREmLN: A Cellular Graph Structure Aware Transcriptomics Foundation Model.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Zero-shot cell-type annotation using pre-trained cellular graph foundation model","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"}],"facts":[{"label":"Recorded dataset","value":"non-immune cells","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"},{"label":"Recorded split","value":"zero-shot","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-54be8a811c206e","kind":"model","name":"AK-score-ensemble","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["akscore-2020"],"links":[],"attributes":{"entity_level":"method","version":"ensemble; learning rate 0.0007","reported_name":"AK-score-ensemble","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"AK-score-ensemble is the method recorded for Protein–ligand binding affinity scoring. This page preserves the configuration reported by AK-Score: Accurate Protein-Ligand Binding Affinity Prediction Using an Ensemble of 3D-Convolutional Neural Networks.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: CASF-2016 scoring-power evaluation.","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column"}],"facts":[{"label":"Recorded dataset","value":"CASF-2016","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column"},{"label":"Recorded configuration","value":"ensemble; learning rate 0.0007","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-54d974e8e08043","kind":"model","name":"mRNA-LM","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["mrna-lm-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"mRNA-LM","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"mRNA-LM is the method recorded for mRNA half-life prediction. This page preserves the configuration reported by mRNA-LM: full-length integrated SLM for mRNA analysis.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: average test performance across cross-validation splits","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"}],"facts":[{"label":"Recorded dataset","value":"mRNA half-life","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"},{"label":"Recorded split","value":"test set across CV splits","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, mRNA-LM row, mRNA half-life Spearman column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-57dbab30462150","kind":"model","name":"CLAPE-SMB with ESM-2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["clape-smb-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"CLAPE-SMB with ESM-2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"CLAPE-SMB with ESM-2 is the method recorded for protein-small molecule binding-site prediction. This page preserves the configuration reported by Protein-small molecule binding site prediction based on a pre-trained protein language model with contrastive learning.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Contrastive CLAPE-SMB binding-site predictor with ESM-2 feature extractor","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"SJC","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, ESM-2 / SJC row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-5b70fccb70bb70","kind":"model","name":"Eco70PromBERT","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["cyaprombert-2022"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Eco70PromBERT","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Eco70PromBERT is the method recorded for E. coli sigma70 promoter prediction. This page preserves the configuration reported by TSSNote-CyaPromBERT: Development of an integrated platform for highly accurate promoter prediction and visualization of Synechococcus sp. and Synechocystis sp. through a state-of-the-art natural language processing model BERT.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: BERT-base with 1bp tokenizer; 110 promoters and 108 non-promoters.","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column"}],"facts":[{"label":"Recorded dataset","value":"Independent E. coli sigma70 test dataset","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column"},{"label":"Recorded split","value":"independent test","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column"}],"coverage":"limited","gaps":["The retained evidence location (TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-60455ff7cc0c15","kind":"model","name":"scRegNet (Geneformer backbone)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scregnet-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scRegNet (Geneformer backbone)","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scRegNet (Geneformer backbone) is the method recorded for Gene-regulatory link prediction. This page preserves the configuration reported by Prediction of Gene Regulatory Connections with Joint Single-Cell Foundation Models and Graph-Based Learning.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TFs plus 500 variable genes; mean from 50 independent evaluations.","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry"}],"facts":[{"label":"Recorded dataset","value":"hESC cell-type-specific GRN","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-62bc5e5ba13e7d","kind":"model","name":"Kraken 2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["lemur-magnet-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Kraken 2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Kraken 2 is the method recorded for Long-read taxonomic profiling. This page preserves the configuration reported by Lightweight taxonomic profiling of long-read metagenomic datasets with Lemur and Magnet.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Mean across five replicate runs on Zymo LOG 10%.","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Kraken 2 row, F1 column"}],"facts":[{"label":"Recorded dataset","value":"Zymo LOG 10%","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Kraken 2 row, F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Kraken 2 row, F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, LOG 10% / Kraken 2 row, F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-65059c3a806306","kind":"model","name":"PlantCAD2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["plantcad2-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"PlantCAD2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"PlantCAD2 is the method recorded for cross-species conservation prediction. This page preserves the configuration reported by PlantCAD2: A Long-Context DNA Language Model for Cross-Species Functional Annotation in Angiosperms.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Zero-shot score for conserved versus non-conserved sites from alignments of 35 Andropogoneae genomes","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"}],"facts":[{"label":"Recorded dataset","value":"Andropogoneae genome-wide conservation","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-673b8f46361000","kind":"model","name":"Kraken2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["lazypipe-2020"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Kraken2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Kraken2 is the method recorded for Simulated metagenome virus-taxon retrieval. This page preserves the configuration reported by Novel NGS pipeline for virus discovery from a wide spectrum of hosts and sample types.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Genus-rank viral taxon retrieval.","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Kraken2 / Genus row, F column"}],"facts":[{"label":"Recorded dataset","value":"Simulated viral metagenome","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Kraken2 / Genus row, F column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Kraken2 / Genus row, F column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, Kraken2 / Genus row, F column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-67ea6bd77b2ed1","kind":"model","name":"ESM2_AMPS","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["esm2-amp-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ESM2_AMPS","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ESM2_AMPS is the method recorded for protein-protein interaction prediction. This page preserves the configuration reported by ESM2_AMP: an interpretable framework for protein–protein interactions prediction and biological mechanism discovery.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: ESM2-derived embeddings plus paper interaction predictor","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"Bernett PPI dataset","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 4, ESM2_AMPS row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-67eaf766fa9877","kind":"model","name":"ICCTax","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["icctax-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ICCTax","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ICCTax is the method recorded for Hierarchical metagenomic taxonomy classification. This page preserves the configuration reported by ICCTax: a hierarchical taxonomic classifier for metagenomic sequences on a large language model.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Macro average precision at genus rank on Complete dataset.","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column"}],"facts":[{"label":"Recorded dataset","value":"ICCTax Complete dataset","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, ICCTax row, Genus column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-688eb780ef7d2e","kind":"model","name":"PC-mer + LR","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["pc-mer-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"PC-mer + LR","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"PC-mer + LR is the method recorded for metagenomic genus classification. This page preserves the configuration reported by PC-mer: An Ultra-fast memory-efficient tool for metagenomics profiling and classification.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: k=8 PC-mer feature extraction with logistic regression on AMP genus-classification dataset","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"}],"facts":[{"label":"Recorded dataset","value":"AMP","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"},{"label":"Recorded split","value":"genus-level","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-6ac0730e8481de","kind":"model","name":"Lemur","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["lemur-magnet-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Lemur","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Lemur is the method recorded for Long-read taxonomic profiling. This page preserves the configuration reported by Lightweight taxonomic profiling of long-read metagenomic datasets with Lemur and Magnet.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Mean across five replicate runs on Zymo LOG 10%.","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column"}],"facts":[{"label":"Recorded dataset","value":"Zymo LOG 10%","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, LOG 10% / Lemur row, F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-6c0bc8d297cc7a","kind":"model","name":"DiffDock-NMDN","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["nmdn-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"DiffDock-NMDN","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DiffDock-NMDN is the method recorded for Protein–ligand virtual screening. This page preserves the configuration reported by Normalized Protein–Ligand Distance Likelihood Score for End-to-End Blind Docking and Virtual Screening.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: NMDN scoring on DiffDock-NMDN blind docked poses; not ligand-pose RMSD.","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column"}],"facts":[{"label":"Recorded dataset","value":"CASF-2016 blind docked poses","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, DiffDock-NMDN / NMDN row, success rate (%) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-6d9dbac97852d8","kind":"model","name":"DETIRE","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["detire-viral-metagenomes-2023"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"DETIRE","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DETIRE is the method recorded for viral sequence detection. This page preserves the configuration reported by DETIRE: a hybrid deep learning model for identifying viral sequences from metagenomes.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Hybrid deep learning virus-fragment classifier on paper testing dataset","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"}],"facts":[{"label":"Recorded dataset","value":"testing viral metagenome dataset","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"},{"label":"Recorded split","value":"test","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, Accuracy row, DETIRE column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-70c732770a200f","kind":"model","name":"Chai-1","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["ibex-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Chai-1","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Chai-1 is the method recorded for Antibody loop structure prediction. This page preserves the configuration reported by Conformation-aware structure prediction of antigen-recognizing immune proteins.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Backbone RMSD after framework alignment; one seed and one diffusion trajectory.","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Chai-1 row, CDR H3 column"}],"facts":[{"label":"Recorded dataset","value":"ImmuneBuilder antibody test set","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Chai-1 row, CDR H3 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Chai-1 row, CDR H3 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, Antibodies / Chai-1 row, CDR H3 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-70f57ebb163a5c","kind":"model","name":"Kraken2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["cammiq-2022"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Kraken2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Kraken2 is the method recorded for Strain-level abundance quantification. This page preserves the configuration reported by Strain level microbial detection and quantification with applications to single cell metagenomics.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Strain-level quantification on the HumanGut-all synthetic query.","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"}],"facts":[{"label":"Recorded dataset","value":"HumanGut-all strain-level query","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-7234658bc9c828","kind":"model","name":"RNAret","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["rnaret-2026"],"links":[],"attributes":{"entity_level":"method","version":"5-mer","reported_name":"RNAret","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"RNAret is the method recorded for miRNA-mRNA interaction prediction. This page preserves the configuration reported by Retentive Network promotes efficient RNA language modeling of long sequences.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: 5-mer RNAret classifier; 72/8/20 train/validation/test split","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"}],"facts":[{"label":"Recorded dataset","value":"MirTarRAW","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"},{"label":"Recorded split","value":"held-out test","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"},{"label":"Recorded configuration","value":"5-mer","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, MirTarRAW / 5-mer RNAret row, F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-73ae07fb5be204","kind":"model","name":"GSMFormer-PPI + ProstT5","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["gsmformer-ppi-2026"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"GSMFormer-PPI + ProstT5","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"GSMFormer-PPI + ProstT5 is the method recorded for protein-protein interaction prediction. This page preserves the configuration reported by Multimodal graph, surface, and language-based model for protein protein interaction prediction.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: ProstT5 embeddings as graph node features","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"paper PPI test set","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"},{"label":"Recorded split","value":"test set","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 6, ProstT5 embedding row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-75e4e5e5965320","kind":"model","name":"DiffDock holo","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["ensemble-idp-docking-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"DiffDock holo","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DiffDock holo is the method recorded for Intrinsically disordered protein ensemble docking. This page preserves the configuration reported by Ensemble docking for intrinsically disordered proteins.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble.","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, DiffDock Holo Docking column"}],"facts":[{"label":"Recorded dataset","value":"α-synuclein Ligand 47 MD ensemble","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, DiffDock Holo Docking column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, DiffDock Holo Docking column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Ligand 47 row, DiffDock Holo Docking column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-77ad27d4098177","kind":"model","name":"scGPT","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scelmo-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scGPT","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scGPT is the method recorded for Cell-type annotation. This page preserves the configuration reported by scELMo: Embeddings from Language Models are Good Learners for Single-cell Data Analysis.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Zero-shot setting; source caption says some comparator rows come from GenePT.","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column"}],"facts":[{"label":"Recorded dataset","value":"hPancreas","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, hPancreas zero-shot / scGPT (z) row, F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-790768ed581685","kind":"model","name":"kMetaShot","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["kmetashot-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"kMetaShot","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"kMetaShot is the method recorded for Mock-community MAG taxonomy classification. This page preserves the configuration reported by kMetaShot: a fast and reliable taxonomy classifier for metagenome-assembled genomes.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Genus classification of MAGs from MegaHIT contigs; uncorrected kMetaShot.","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column"}],"facts":[{"label":"Recorded dataset","value":"Real mock community MAGs","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, F1-score % row, Genus kMS column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-7ad28cd57f5f5b","kind":"model","name":"ERNIE-RNA + CoBRA","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["cobra-rna-binding-2026"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"ERNIE-RNA + CoBRA","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ERNIE-RNA + CoBRA is the method recorded for RNA compound-binding site prediction. This page preserves the configuration reported by CoBRA: compound binding site prediction using RNA language model.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: ERNIE-RNA embedding with TCL focal loss","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"}],"facts":[{"label":"Recorded dataset","value":"CoBRA compound-binding test set","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"},{"label":"Recorded split","value":"test set","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, ERNIE-RNA / TCL focal row, MCC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-7b052acf17b5ba","kind":"model","name":"NCD-gzip","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["ncd-metagenomics-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"NCD-gzip","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"NCD-gzip is the method recorded for CAMI II superkingdom read classification. This page preserves the configuration reported by Normalized compression distance for DNA classification.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Superkingdom-level macro-averaged F1; NCD assigns every read. Phylum-level macro-averaged F1; distinct taxonomic rank from the other row.","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column; Table 5, NCD Phylum row, F1 column"}],"facts":[{"label":"Recorded dataset","value":"CAMI II Sample_0 10,000-read subsample","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column; Table 5, NCD Phylum row, F1 column"},{"label":"Recorded dataset","value":"CAMI II Sample_0 10,000-read subsample","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column; Table 5, NCD Phylum row, F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column; Table 5, NCD Phylum row, F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, NCD Superkingdom row, F1 column; Table 5, NCD Phylum row, F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-7c595040de69bc","kind":"model","name":"GenePT-w","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["genept-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"GenePT-w","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"GenePT-w is the method recorded for Cell-type structure in frozen embeddings. This page preserves the configuration reported by GenePT: A Simple But Effective Foundation Model for Genes and Cells Built From ChatGPT.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: k-means on pretrained cell embeddings; agreement with original cell-type labels.","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column"}],"facts":[{"label":"Recorded dataset","value":"Aorta single-cell dataset","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Aorta / Cell type row, GenePT-w ARI column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-7cf2f9951e1dba","kind":"model","name":"scaLR","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scalr-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scaLR","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scaLR is the method recorded for PBMC cell-type classification. This page preserves the configuration reported by scaLR: a low-resource deep neural network-based platform for single cell analysis and biomarker discovery.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: All features and samples from PBMCs-BS.","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column"}],"facts":[{"label":"Recorded dataset","value":"PBMCs-BS","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, scaLR row, Cell type Accuracy column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-7dd5992188a868","kind":"model","name":"PST","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["pst-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"PST","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"PST is the method recorded for Zero-shot variant effect prediction. This page preserves the configuration reported by Endowing protein language models with structural knowledge.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Zero-shot VEP; paper averages absolute Spearman correlations.","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column"}],"facts":[{"label":"Recorded dataset","value":"ProteinShake VEP datasets","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, PST row, Zero-shot VEP Mean |ρ| column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-7f1165b35f10e2","kind":"model","name":"DNABERT-2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["genomeocean-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"DNABERT-2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DNABERT-2 is the method recorded for Natural vs artificial microbial genome sequence. This page preserves the configuration reported by GenomeOcean: An Efficient Genome Foundation Model Trained on Large-Scale Metagenomic Assemblies.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Source reports natural-versus-artificial sequence classification.","source_ids":["genomeocean-2025"],"source_locator":"Table 2, DNABERT-2 row, F1 column"}],"facts":[{"label":"Recorded dataset","value":"GenomeOcean natural/artificial sequence test","source_ids":["genomeocean-2025"],"source_locator":"Table 2, DNABERT-2 row, F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["genomeocean-2025"],"source_locator":"Table 2, DNABERT-2 row, F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, DNABERT-2 row, F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-7f5b8234967c54","kind":"model","name":"2OMe-LM","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["2ome-lm-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"2OMe-LM","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"2OMe-LM is the method recorded for human RNA 2-prime-O-methylation site prediction. This page preserves the configuration reported by 2OMe-LM: predicting 2′-O-methylation sites in human RNA using a pre-trained RNA language model.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: pretrained RNA language model predictor","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"}],"facts":[{"label":"Recorded dataset","value":"human RNA 2OMe sites","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"},{"label":"Recorded split","value":"5-fold cross-validation","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, 2OMe-LM row, AUC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-7f6ffd9e2a08be","kind":"model","name":"DiffDock","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["boltz-stereochemistry-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"DiffDock","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DiffDock is the method recorded for Protein–ligand pose prediction. This page preserves the configuration reported by Improving Stereochemical Limitations in Protein–Ligand Complex Structure Prediction.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: All entries; rigid-protein docking comparator; authors note this dataset contains structures seen during model training.","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, DiffDock row, Ligand RMSD (Å) column"}],"facts":[{"label":"Recorded dataset","value":"PLINDER-L95","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, DiffDock row, Ligand RMSD (Å) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, DiffDock row, Ligand RMSD (Å) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, DiffDock row, Ligand RMSD (Å) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-808b23c65fbc89","kind":"model","name":"Cell-DINO ViT-L","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["cell-dino-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Cell-DINO ViT-L","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Cell-DINO ViT-L is the method recorded for protein localization classification. This page preserves the configuration reported by Cell-DINO: Self-supervised image-based embeddings for cell fluorescent microscopy.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Self-supervised microscopy embedding pre-trained on HPA-FoV; downstream protein-localization classifier. Dataset-specific pretraining; the paper does not claim a general-purpose foundation model that generalizes beyond these benchmarks.","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"}],"facts":[{"label":"Recorded dataset","value":"HPA-FoV","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, HPA-FoV section, Cell-DINO row, PL column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-8100b3de6c7811","kind":"model","name":"PMF (LASSO)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["fingerprint-scoring-2022"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"PMF (LASSO)","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"PMF (LASSO) is the method recorded for Protein–ligand binding energy prediction. This page preserves the configuration reported by Machine-Learning- and Knowledge-Based Scoring Functions Incorporating Ligand and Protein Fingerprints.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: PMF-only LASSO baseline evaluated by the same authors.","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF / LASSO row, R column"}],"facts":[{"label":"Recorded dataset","value":"Fingerprint-scoring benchmark","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF / LASSO row, R column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF / LASSO row, R column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, PMF / LASSO row, R column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-81b0394d5ac3e8","kind":"model","name":"TransBind","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["transbind-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"TransBind","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"TransBind is the method recorded for transcription-factor DNA binding-site prediction. This page preserves the configuration reported by Integrating protein and DNA embeddings for improving genome-wide transcription factor binding site prediction.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Integrates protein and DNA embeddings for TFBS prediction on paper test dataset","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"genome-wide TF binding sites","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"},{"label":"Recorded split","value":"test","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, TransBind row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-86393c76dd8fa9","kind":"model","name":"MegSite + ESM3","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["megsite-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"MegSite + ESM3","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"MegSite + ESM3 is the method recorded for DNA-binding residue prediction. This page preserves the configuration reported by MegSite: an accurate nucleic acid-binding residue prediction method based on multimodal protein language model.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: ESM3 multimodal embedding ablation in MegSite","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"}],"facts":[{"label":"Recorded dataset","value":"DNA-129_Test","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"},{"label":"Recorded split","value":"independent test","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, DNA-129_Test / ESM3 row, AUC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-884582fb0c70dc","kind":"model","name":"Boltz-1","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["antibody-flexibility-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Boltz-1","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Boltz-1 is the method recorded for Antibody–antigen interaction prediction using folded complexes. This page preserves the configuration reported by Enhancing antibody-antigen interaction prediction with atomic flexibility.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Interaction classifier evaluated using Boltz-1-folded input complexes; this is pipeline AUC, not DockQ.","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Boltz-1 (no MSA) column"}],"facts":[{"label":"Recorded dataset","value":"Antibody–antigen GEP test set","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Boltz-1 (no MSA) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Boltz-1 (no MSA) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, Folded row, Boltz-1 (no MSA) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-8861b9ad9b9c9b","kind":"model","name":"Kraken2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["icctax-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Kraken2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Kraken2 is the method recorded for Hierarchical metagenomic taxonomy classification. This page preserves the configuration reported by ICCTax: a hierarchical taxonomic classifier for metagenomic sequences on a large language model.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Macro average precision at genus rank on Complete dataset.","source_ids":["icctax-2025"],"source_locator":"Table 2, Kraken2 row, Genus column"}],"facts":[{"label":"Recorded dataset","value":"ICCTax Complete dataset","source_ids":["icctax-2025"],"source_locator":"Table 2, Kraken2 row, Genus column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["icctax-2025"],"source_locator":"Table 2, Kraken2 row, Genus column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Kraken2 row, Genus column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-89f5a8f309fa18","kind":"model","name":"scRegNet (scBERT backbone)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scregnet-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scRegNet (scBERT backbone)","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scRegNet (scBERT backbone) is the method recorded for Gene-regulatory link prediction. This page preserves the configuration reported by Prediction of Gene Regulatory Connections with Joint Single-Cell Foundation Models and Graph-Based Learning.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: TFs plus 500 variable genes; mean from 50 independent evaluations.","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"}],"facts":[{"label":"Recorded dataset","value":"hESC cell-type-specific GRN","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-8ad3e0cefde796","kind":"model","name":"Vaxign-DL + ESM","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["vaxign-esm-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Vaxign-DL + ESM","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Vaxign-DL + ESM is the method recorded for vaccine-antigen candidate prediction. This page preserves the configuration reported by Enhancing Vaxign-DL for Vaccine Candidate Prediction with added ESM-Generated Features.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Combined skip architecture, four layers, ESM-generated sequence features","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"}],"facts":[{"label":"Recorded dataset","value":"vaccine candidate validation","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, 4 Layers row, AUPRC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-8cb3dd4e9f5b10","kind":"model","name":"Lazypipe-nt","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["lazypipe-2020"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Lazypipe-nt","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Lazypipe-nt is the method recorded for Simulated metagenome virus-taxon retrieval. This page preserves the configuration reported by Novel NGS pipeline for virus discovery from a wide spectrum of hosts and sample types.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Genus-rank viral taxon retrieval.","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column"}],"facts":[{"label":"Recorded dataset","value":"Simulated viral metagenome","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, Lazypipe-nt / Genus row, F column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-8d2c291733dfe1","kind":"model","name":"DeepInterAware","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["deepinteraware-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"DeepInterAware","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DeepInterAware is the method recorded for antigen-antibody HIV neutralization prediction. This page preserves the configuration reported by DeepInterAware: Deep Interaction Interface‐Aware Network for Improving Antigen‐Antibody Interaction Prediction from Sequence Data.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Sequence-based interface-aware model, antibody-unseen split","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"HIV neutralization","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"},{"label":"Recorded split","value":"antibody-unseen","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Ab Unseen section, DeepInterAware row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-8db190bee6aae5","kind":"model","name":"ENBED","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["enbed-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ENBED","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ENBED is the method recorded for Enhancer classification. This page preserves the configuration reported by Understanding the natural language of DNA using encoder–decoder foundation models with byte-level precision.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Reported Genomic Benchmarks classification accuracy.","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column"}],"facts":[{"label":"Recorded dataset","value":"Genomic Benchmarks Mouse Enhancers","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Mouse Enhancers row, ENBED column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-953007693fb72a","kind":"model","name":"HyenaDNA","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["polya-glm-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"HyenaDNA","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"HyenaDNA is the method recorded for polyadenylation site detection. This page preserves the configuration reported by PolyA-GLM: A comprehensive framework for De novo polyadenylation site prediction using genome language models.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: few-shot Gene-Gene negative-set comparison","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"}],"facts":[{"label":"Recorded dataset","value":"poly(A) Gene-Gene","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"},{"label":"Recorded split","value":"5-fold cross-validation","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, Few-shot HyenaDNA row, G-G AUC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-9a200c55b0e03e","kind":"model","name":"iPro-MP","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["ipromp-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"iPro-MP","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"iPro-MP is the method recorded for Multi-species prokaryotic promoter detection. This page preserves the configuration reported by iPro-MP: a BERT-based model to predict multiple prokaryotic promoters.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Average over independent testing sets.","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column"}],"facts":[{"label":"Recorded dataset","value":"23 independent prokaryotic promoter test sets","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column"},{"label":"Recorded split","value":"independent test","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, iPro-MP row, AUC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-9b3bc255532dd3","kind":"model","name":"HyenaDNA","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["dnalongbench-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"HyenaDNA","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"HyenaDNA is the method recorded for Enhancer-target gene prediction. This page preserves the configuration reported by DNALongBench: A Benchmark Suite for Long-Range DNA Prediction Tasks.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Long-range ETGP benchmark; source table reports AUROC.","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column"}],"facts":[{"label":"Recorded dataset","value":"DNALongBench ETGP","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, HyenaDNA row, ETGP column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-9c10fbec02a365","kind":"model","name":"FUJISAN","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["fujisan-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"FUJISAN","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"FUJISAN is the method recorded for Enzyme functional identity prediction. This page preserves the configuration reported by Enhanced prediction of protein functional identity through the integration of sequence and structural features.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Sequence and structural feature integration; paper-reported test sub-dataset.","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"FUJISAN test sub-dataset","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, FUJISAN row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-9f39de53f7a139","kind":"model","name":"scXDR","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scxdr-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scXDR","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scXDR is the method recorded for Cross-dataset single-cell drug response transfer. This page preserves the configuration reported by scXDR: drug response prediction across single-cell datasets via heterogeneous network transfer learning.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Single-cell-to-single-cell transfer; source scenario 2.","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column"}],"facts":[{"label":"Recorded dataset","value":"scXDR transfer scenario 2","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, scXDR row, Scenario 2 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-a0db32ae53e5ed","kind":"model","name":"scLLMDA","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scatac-llmda-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scLLMDA","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scLLMDA is the method recorded for Cross-platform scATAC cell-type annotation. This page preserves the configuration reported by Cell type annotation for scATAC-seq via DNA large language model and graph domain adaptation.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Cross-platform reference-query cell-type annotation.","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column"}],"facts":[{"label":"Recorded dataset","value":"MosA1 reference → WholeBrainA query","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-a29203c09857ef","kind":"model","name":"MULAN-ESM2 S","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["mulan-2025"],"links":[],"attributes":{"entity_level":"method","version":"small ESM2 backbone","reported_name":"MULAN-ESM2 S","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"MULAN-ESM2 S is the method recorded for human protein-protein interaction prediction. This page preserves the configuration reported by MULAN: multimodal protein language model for sequence and structure encoding.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: MULAN sequence-structure model based on ESM2 8M","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"}],"facts":[{"label":"Recorded dataset","value":"HumanPPI","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"},{"label":"Recorded split","value":"paper evaluation","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"},{"label":"Recorded configuration","value":"small ESM2 backbone","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, MULAN-ESM2 S row, HumanPPI AUC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-a7cfacf25d97ad","kind":"model","name":"Caduceus-Ph","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["dnalongbench-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Caduceus-Ph","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Caduceus-Ph is the method recorded for Enhancer-target gene prediction. This page preserves the configuration reported by DNALongBench: A Benchmark Suite for Long-Range DNA Prediction Tasks.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Long-range ETGP benchmark; source table reports AUROC.","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, Caduceus-Ph row, ETGP column"}],"facts":[{"label":"Recorded dataset","value":"DNALongBench ETGP","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, Caduceus-Ph row, ETGP column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, Caduceus-Ph row, ETGP column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, Caduceus-Ph row, ETGP column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-aa763db2cfdeff","kind":"model","name":"EVO2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["lambda-prophage-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"EVO2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"EVO2 is the method recorded for Genome-wide prophage detection. This page preserves the configuration reported by LAMBDA: A Prophage Detection Benchmark for Genomic Language Models.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Genomic language model fine-tuned for prophage detection; genome-wide evaluation.","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column"}],"facts":[{"label":"Recorded dataset","value":"LAMBDA genome-wide prophage test","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, EVO2 row, MCC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-ab02228f50a37c","kind":"model","name":"C2S (GPT-2 Large)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["cell2sentence-2024"],"links":[],"attributes":{"entity_level":"method","version":"GPT-2 Large","reported_name":"C2S (GPT-2 Large)","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"C2S (GPT-2 Large) is the method recorded for Combinatorial cell-label classification. This page preserves the configuration reported by Cell2Sentence: Teaching Large Language Models the Language of Biology.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Partial-credit labels including cell type, perturbation, and dose.","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column"}],"facts":[{"label":"Recorded dataset","value":"L1000","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column"},{"label":"Recorded configuration","value":"GPT-2 Large","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-abc19288fe9009","kind":"model","name":"Caduceus","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[],"attributes":{"entity_level":"method","version":"8M","reported_name":"Caduceus","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Caduceus is the method recorded for G-quadruplex classification. This page preserves the configuration reported by Benchmarking DNA large language models on quadruplexes.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Pretrained model evaluated on KEx as reported in Table 5.","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, Caduceus (8 M) row, Accuracy column"}],"facts":[{"label":"Recorded dataset","value":"KEx","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, Caduceus (8 M) row, Accuracy column"},{"label":"Recorded configuration","value":"8M","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, Caduceus (8 M) row, Accuracy column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, Caduceus (8 M) row, Accuracy column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, Caduceus (8 M) row, Accuracy column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-ade36035f58f27","kind":"model","name":"DNABERT-2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[],"attributes":{"entity_level":"method","version":"117M","reported_name":"DNABERT-2","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DNABERT-2 is the method recorded for G-quadruplex classification. This page preserves the configuration reported by Benchmarking DNA large language models on quadruplexes.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Pretrained model evaluated on KEx as reported in Table 5.","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column"}],"facts":[{"label":"Recorded dataset","value":"KEx","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column"},{"label":"Recorded configuration","value":"117M","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, DNABERT-2 (117 M) row, Accuracy column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-b3fdf259d51533","kind":"model","name":"Chai-1","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["antibody-flexibility-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Chai-1","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Chai-1 is the method recorded for Antibody–antigen interaction prediction using folded complexes. This page preserves the configuration reported by Enhancing antibody-antigen interaction prediction with atomic flexibility.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Interaction classifier evaluated using Chai-1-folded input complexes; this is pipeline AUC, not DockQ.","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column"}],"facts":[{"label":"Recorded dataset","value":"Antibody–antigen GEP test set","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, Folded row, Chai-1 (no MSA) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-b46ae14b9927ac","kind":"model","name":"Geneformer","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["single-cell-peft-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Geneformer","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Geneformer is the method recorded for Cell-type identification. This page preserves the configuration reported by Parameter-Efficient Fine-Tuning Enhances Adaptation of Single Cell Large Language Model for Cell Type Identification.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Native scLLM cell-type identification as reported in Table 2.","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / Geneformer row, F1-Score column"}],"facts":[{"label":"Recorded dataset","value":"M.S. single-cell dataset","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / Geneformer row, F1-Score column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / Geneformer row, F1-Score column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, M.S. / Geneformer row, F1-Score column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-bdb1db16d3389d","kind":"model","name":"ENBED (GRCh38)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["enbed-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ENBED (GRCh38)","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ENBED (GRCh38) is the method recorded for Enhancer classification. This page preserves the configuration reported by Understanding the natural language of DNA using encoder–decoder foundation models with byte-level precision.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: ENBED trained on GRCh38; reported Genomic Benchmarks classification accuracy.","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column"}],"facts":[{"label":"Recorded dataset","value":"Genomic Benchmarks Mouse Enhancers","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED (GRCh38) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Mouse Enhancers row, ENBED (GRCh38) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-c464bface507ee","kind":"model","name":"BPfold","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["bpfold-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"BPfold","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"BPfold is the method recorded for RNA secondary structure. This page preserves the configuration reported by Deep generalizable prediction of RNA secondary structure via base pair motif energy.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Family-wise evaluation of canonical base-pair predictions.","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column"}],"facts":[{"label":"Recorded dataset","value":"PDB RNA set","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, BPfold row, PDB F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-cbbe04b826ceff","kind":"model","name":"PhyloGPN","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["phylogpn-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"PhyloGPN","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"PhyloGPN is the method recorded for ClinVar 3-prime UTR variant classification. This page preserves the configuration reported by A Phylogenetic Approach to Genomic Language Modeling.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: log-likelihood-ratio scoring","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"}],"facts":[{"label":"Recorded dataset","value":"ClinVar 3-prime UTR variants","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"},{"label":"Recorded split","value":"paper evaluation","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, 3-prime UTR row, PhyloGPN AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-ccd1160ad4ec27","kind":"model","name":"ESM2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["fujisan-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ESM2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ESM2 is the method recorded for Enzyme functional identity prediction. This page preserves the configuration reported by Enhanced prediction of protein functional identity through the integration of sequence and structural features.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Comparator evaluated on the paper-reported test sub-dataset.","source_ids":["fujisan-2024"],"source_locator":"Table 1, ESM2 row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"FUJISAN test sub-dataset","source_ids":["fujisan-2024"],"source_locator":"Table 1, ESM2 row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["fujisan-2024"],"source_locator":"Table 1, ESM2 row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, ESM2 row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-cd246741c378db","kind":"model","name":"CodonBERT","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["codonbert-vaccines-2024"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"CodonBERT","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"CodonBERT is the method recorded for flu-vaccine mRNA property prediction. This page preserves the configuration reported by CodonBERT large language model for mRNA vaccines.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: codon-based model fine-tuned for downstream regression","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"}],"facts":[{"label":"Recorded dataset","value":"flu-vaccine sequences","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"},{"label":"Recorded split","value":"paper evaluation","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, CodonBERT row, Flu vaccines Spearman correlation column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-cdc9aabf4efc04","kind":"model","name":"Boltz-2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["mpro-pose-affinity-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Boltz-2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Boltz-2 is the method recorded for Ligand potency prediction using generated poses. This page preserves the configuration reported by A Comparative Study of Deep Learning and Classical Modeling Approaches for Protein–Ligand Binding Pose and Affinity Prediction in Coronavirus Main Proteases.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Potency prediction using Boltz-2 ligand-pose generation protocol; see paper scoring pipeline.","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column"}],"facts":[{"label":"Recorded dataset","value":"SARS-CoV-2 Mpro ligands","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, Boltz-2 row, Pearson’s R column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d023fbe78bc4df","kind":"model","name":"ERNIE-RNA","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["ernie-rna-2025"],"links":[],"attributes":{"entity_level":"method","version":"86M","reported_name":"ERNIE-RNA","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ERNIE-RNA is the method recorded for RNA secondary-structure prediction. This page preserves the configuration reported by ERNIE-RNA: an RNA language model with structure-enhanced representations.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: zero-shot attention-derived base-pair prediction","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"}],"facts":[{"label":"Recorded dataset","value":"bpRNA-new","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"},{"label":"Recorded split","value":"cross-family test","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"},{"label":"Recorded configuration","value":"86M","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d0677d52d2b9fd","kind":"model","name":"geNomad","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["lambda-prophage-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"geNomad","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"geNomad is the method recorded for Genome-wide prophage detection. This page preserves the configuration reported by LAMBDA: A Prophage Detection Benchmark for Genomic Language Models.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Traditional specialist comparator; genome-wide evaluation.","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, geNomad row, MCC column"}],"facts":[{"label":"Recorded dataset","value":"LAMBDA genome-wide prophage test","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, geNomad row, MCC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, geNomad row, MCC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 5, geNomad row, MCC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d0d5df2beb02b2","kind":"model","name":"ESM-2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["prime-2026"],"links":[],"attributes":{"entity_level":"method","version":"8M","reported_name":"ESM-2","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ESM-2 is the method recorded for Mutated RBD binding prediction. This page preserves the configuration reported by PRIME: An evaluation framework for protein representation inference and generalization in viral mutation space.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Frozen mean-pooled representation with downstream regression; position-stratified split.","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"}],"facts":[{"label":"Recorded dataset","value":"PRIME mutated RBD","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"},{"label":"Recorded split","value":"position-stratified","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"},{"label":"Recorded configuration","value":"8M","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d0e594ec3c0430","kind":"model","name":"DNABERT2-Enhancer","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["dnabert2-enhancer-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"DNABERT2-Enhancer","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DNABERT2-Enhancer is the method recorded for enhancer recognition. This page preserves the configuration reported by Utilizing a deep learning model based on BERT for identifying enhancers and their strength.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: first-layer enhancer versus non-enhancer classifier","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"}],"facts":[{"label":"Recorded dataset","value":"Liu training dataset","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"},{"label":"Recorded split","value":"5-fold cross-validation","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 4, first-layer DNABERT2-Enhancer row, AUC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d1cd9a425f9bbd","kind":"model","name":"TU-Fold (aug)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["tu-fold-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"TU-Fold (aug)","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"TU-Fold (aug) is the method recorded for RNA secondary structure. This page preserves the configuration reported by RNA secondary structure prediction by conducting multi-class classifications.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Three-fold training and evaluation; source reports mean and standard deviation.","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column"}],"facts":[{"label":"Recorded dataset","value":"RNA8F","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, TU-Fold (aug) row, Overall F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d25dab1a9c4fff","kind":"model","name":"ESM-2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["pst-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ESM-2","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ESM-2 is the method recorded for Zero-shot variant effect prediction. This page preserves the configuration reported by Endowing protein language models with structural knowledge.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Zero-shot VEP; paper averages absolute Spearman correlations.","source_ids":["pst-2025"],"source_locator":"Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"}],"facts":[{"label":"Recorded dataset","value":"ProteinShake VEP datasets","source_ids":["pst-2025"],"source_locator":"Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["pst-2025"],"source_locator":"Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d2c81acf1c42c4","kind":"model","name":"position-aware CNN","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["enhancer-position-encoding-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"position-aware CNN","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"position-aware CNN is the method recorded for enhancer prediction. This page preserves the configuration reported by A deep learning model for DNA enhancer prediction based on nucleotide position aware feature encoding.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Nucleotide position-aware feature encoding; average assessment of CNN classifier","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"}],"facts":[{"label":"Recorded dataset","value":"human enhancer dataset","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Human section, CNN row, AUC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d326e3c4e3ba20","kind":"model","name":"ESM-2","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["proteingym-2023"],"links":[],"attributes":{"entity_level":"method","version":"15B","reported_name":"ESM-2","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ESM-2 is the method recorded for Zero-shot substitution mutation effects: stability. This page preserves the configuration reported by ProteinGym: Large-Scale Benchmarks for Protein Design and Fitness Prediction.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Zero-shot mutation scores; average Spearman across stability-category assays.","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column"}],"facts":[{"label":"Recorded dataset","value":"ProteinGym substitution DMS: stability assays","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column"},{"label":"Recorded configuration","value":"15B","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column"}],"coverage":"limited","gaps":["The retained evidence location (Table A7, ESM-2 (15B) row, Stability column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d3fd83835a2d44","kind":"model","name":"BiRNA-BERT","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["birna-bert-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"BiRNA-BERT","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"BiRNA-BERT is the method recorded for extremely long RNA species classification. This page preserves the configuration reported by BiRNA-BERT allows efficient RNA language modeling with adaptive tokenization.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: adaptive tokenization on full-length long RNA sequences","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"}],"facts":[{"label":"Recorded dataset","value":"extremely long-sequence species classification","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"},{"label":"Recorded split","value":"paper evaluation","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, BiRNA-BERT row, F1 Score column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d5bc536ca6f0d3","kind":"model","name":"Caduceus (character tokens)","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[],"attributes":{"entity_level":"method","version":"3.9M parameter variant","reported_name":"Caduceus (character tokens)","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Caduceus (character tokens) is the method recorded for regulatory sequence classification. This page preserves the configuration reported by The impact of tokenizer selection in genomic language models.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: task-category MCC across benchmark datasets","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"}],"facts":[{"label":"Recorded dataset","value":"genomic benchmark categories","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"},{"label":"Recorded split","value":"paper benchmark summary","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"},{"label":"Recorded configuration","value":"3.9M parameter variant","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Regulatory row, Caduceus (char) MCC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d60f505aabb19c","kind":"model","name":"Geneformer","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scelmo-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Geneformer","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Geneformer is the method recorded for Cell-type annotation. This page preserves the configuration reported by scELMo: Embeddings from Language Models are Good Learners for Single-cell Data Analysis.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Zero-shot setting; source caption says some comparator rows come from GenePT.","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"}],"facts":[{"label":"Recorded dataset","value":"hPancreas","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d6a7fa854437e8","kind":"model","name":"scGPT + residual geometry","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["single-cell-residual-geometry-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scGPT + residual geometry","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scGPT + residual geometry is the method recorded for gene-regulatory signal prediction. This page preserves the configuration reported by Residual-stream geometry of single-cell foundation models carries incremental gene-regulatory signal across tissues.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Asymmetric extraction, PCA-64 centered cosine geometry added to scGPT baseline","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"}],"facts":[{"label":"Recorded dataset","value":"immune tissue","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 4, Immune row, scGPT > +geom AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-d9a06805b36b8a","kind":"model","name":"Boltz-1","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["boltz-stereochemistry-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Boltz-1","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Boltz-1 is the method recorded for Protein–ligand pose prediction. This page preserves the configuration reported by Improving Stereochemical Limitations in Protein–Ligand Complex Structure Prediction.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: All entries; authors note this dataset contains structures seen during model training.","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column"}],"facts":[{"label":"Recorded dataset","value":"PLINDER-L95","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, Boltz-1 row, Ligand RMSD (Å) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-de89576d8d316b","kind":"model","name":"AK-score-single","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["akscore-2020"],"links":[],"attributes":{"entity_level":"method","version":"single; learning rate 0.0007","reported_name":"AK-score-single","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"AK-score-single is the method recorded for Protein–ligand binding affinity scoring. This page preserves the configuration reported by AK-Score: Accurate Protein-Ligand Binding Affinity Prediction Using an Ensemble of 3D-Convolutional Neural Networks.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: CASF-2016 scoring-power evaluation.","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"}],"facts":[{"label":"Recorded dataset","value":"CASF-2016","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"},{"label":"Recorded configuration","value":"single; learning rate 0.0007","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-df0efcc0346224","kind":"model","name":"binding-affinity meta-model","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["ligand-affinity-meta-model-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"binding-affinity meta-model","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"binding-affinity meta-model is the method recorded for protein-ligand binding affinity prediction. This page preserves the configuration reported by Improved Prediction of Ligand–Protein Binding Affinities by Meta-modeling.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Sequence-or-structure meta-model; predicts ln(Kd/Ki) using docked and deep-learning components","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"}],"facts":[{"label":"Recorded dataset","value":"CASF-2016","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"},{"label":"Recorded split","value":"core benchmark","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 4, Meta-models row, CASF-2016 Benchmark > PCC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-df4084611520b7","kind":"model","name":"MetaPhlAn3","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["nabas-plus-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"MetaPhlAn3","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"MetaPhlAn3 is the method recorded for Metagenomic taxonomic classification. This page preserves the configuration reported by Advancing metagenomic classification with NABAS+: a novel alignment-based approach.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Newly generated sample19 used for classifier comparison.","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"}],"facts":[{"label":"Recorded dataset","value":"CAMI II Toy human gastrooral sample19-new","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, Sample19-new / MetaPhlAn3 row, F1 score column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-e0443048c6e110","kind":"model","name":"ProteinMPNN","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["proteingym-2023"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ProteinMPNN","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ProteinMPNN is the method recorded for Zero-shot substitution mutation effects: stability. This page preserves the configuration reported by ProteinGym: Large-Scale Benchmarks for Protein Design and Fitness Prediction.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Zero-shot mutation scores; average Spearman across stability-category assays.","source_ids":["proteingym-2023"],"source_locator":"Table A7, ProteinMPNN row, Stability column"}],"facts":[{"label":"Recorded dataset","value":"ProteinGym substitution DMS: stability assays","source_ids":["proteingym-2023"],"source_locator":"Table A7, ProteinMPNN row, Stability column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["proteingym-2023"],"source_locator":"Table A7, ProteinMPNN row, Stability column"}],"coverage":"limited","gaps":["The retained evidence location (Table A7, ProteinMPNN row, Stability column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-e2f2f0d4830bb0","kind":"model","name":"Mouse-Geneformer","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["mouse-geneformer-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Mouse-Geneformer","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Mouse-Geneformer is the method recorded for Human thymus cell-type classification. This page preserves the configuration reported by Mouse-Geneformer: A deep learning model for mouse single-cell transcriptome and its cross-species utility.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Ortholog-based gene conversion; zero-shot mouse model on human cells.","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column"}],"facts":[{"label":"Recorded dataset","value":"Human thymus scRNA-seq","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-e3abb0b9a2ec79","kind":"model","name":"UFold","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["tu-fold-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"UFold","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"UFold is the method recorded for RNA secondary structure. This page preserves the configuration reported by RNA secondary structure prediction by conducting multi-class classifications.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Three-fold training and evaluation; source reports mean and standard deviation.","source_ids":["tu-fold-2025"],"source_locator":"Table 2, UFold row, Overall F1 column"}],"facts":[{"label":"Recorded dataset","value":"RNA8F","source_ids":["tu-fold-2025"],"source_locator":"Table 2, UFold row, Overall F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["tu-fold-2025"],"source_locator":"Table 2, UFold row, Overall F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, UFold row, Overall F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-e4710b1c3facf2","kind":"model","name":"ESM-2 embedding + paper classifier","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["clathrin-plm-2025"],"links":[],"attributes":{"entity_level":"method","version":"not stated in table","reported_name":"ESM-2 embedding + paper classifier","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ESM-2 embedding + paper classifier is the method recorded for clathrin protein classification. This page preserves the configuration reported by Advancing the accuracy of clathrin protein prediction through multi-source protein language models.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: single-feature ESM-2 embedding comparison","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"}],"facts":[{"label":"Recorded dataset","value":"CLA-IND0.6","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"},{"label":"Recorded split","value":"independent test","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"},{"label":"Recorded configuration","value":"not stated in table","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Independent test / ESM-2 row, ACC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-e6ba198c2ac996","kind":"model","name":"MDL4Microbiome","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["mdl4microbiome-2022"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"MDL4Microbiome","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"MDL4Microbiome is the method recorded for microbiome disease-state classification. This page preserves the configuration reported by Multimodal deep learning applied to classify healthy and disease states of human microbiome.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Multimodal deep learning model on colorectal-cancer versus healthy microbiome samples","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"}],"facts":[{"label":"Recorded dataset","value":"CRC microbiome cohort","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, CRC row, MDL4Microbiome column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-e78e3886df0d3a","kind":"model","name":"VirSorter","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["viral-contig-simulation-2021"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"VirSorter","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"VirSorter is the method recorded for Simulated prophage-contig detection. This page preserves the configuration reported by Simulation study and comparative evaluation of viral contiguous sequence identification tools.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Average across twenty medium- and high-complexity simulated communities.","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, VirSorter row, Prophage F1 column"}],"facts":[{"label":"Recorded dataset","value":"20 medium/high-complexity viral simulations","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, VirSorter row, Prophage F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, VirSorter row, Prophage F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, VirSorter row, Prophage F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-e7d203bd99ca99","kind":"model","name":"NABAS+","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["microbes-communities"]},"source_ids":["nabas-plus-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"NABAS+","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"NABAS+ is the method recorded for Metagenomic taxonomic classification. This page preserves the configuration reported by Advancing metagenomic classification with NABAS+: a novel alignment-based approach.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Newly generated sample19 used for classifier comparison.","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column"}],"facts":[{"label":"Recorded dataset","value":"CAMI II Toy human gastrooral sample19-new","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, Sample19-new / NABAS+ row, F1 score column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-eae60780097101","kind":"model","name":"Chai-1","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["lipp-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"Chai-1","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"Chai-1 is the method recorded for Lipid–protein binding pose. This page preserves the configuration reported by The LiPP Benchmark Set for Modeling Lipid–Protein Complexes: Comparison of Co-Folding and Docking Methods.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Top-scoring pose; all-atom lipid RMSD below 2 Å.","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column"}],"facts":[{"label":"Recorded dataset","value":"LiPP lipid–protein complexes","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Chai-1 row, LiPP (N=331) % Success Rate column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-ed7f0db85facb1","kind":"model","name":"RNAfold","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["rna-transcriptomes"]},"source_ids":["debfold-2024"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"RNAfold","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"RNAfold is the method recorded for RNA secondary structure. This page preserves the configuration reported by DEBFold: Computational Identification of RNA Secondary Structures for Sequences across Structural Families Using Deep Learning.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Median F1 on the prepared TestSetβ.","source_ids":["debfold-2024"],"source_locator":"Table 1, RNAfold row, TestSetβ F1 (%) column"}],"facts":[{"label":"Recorded dataset","value":"DEBFold TestSetβ","source_ids":["debfold-2024"],"source_locator":"Table 1, RNAfold row, TestSetβ F1 (%) column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["debfold-2024"],"source_locator":"Table 1, RNAfold row, TestSetβ F1 (%) column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, RNAfold row, TestSetβ F1 (%) column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-ee1ae8162c7d67","kind":"model","name":"ARSENAL+ChromBPNet","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["dna-genomes"]},"source_ids":["arsenal-regulatory-dna-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ARSENAL+ChromBPNet","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ARSENAL+ChromBPNet is the method recorded for regulatory-variant scoring. This page preserves the configuration reported by Short-Context Regulatory DNA Language Models with Motif-Discovery Regularization.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Supervised ChromBPNet variant scoring with ARSENAL motif-discovery regularization","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"Yoruban LCL dsQTLs","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-f0c630d0565e64","kind":"model","name":"MINGLE","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scatac-llmda-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"MINGLE","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"MINGLE is the method recorded for Cross-platform scATAC cell-type annotation. This page preserves the configuration reported by Cell type annotation for scATAC-seq via DNA large language model and graph domain adaptation.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Cross-platform reference-query comparator.","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"}],"facts":[{"label":"Recorded dataset","value":"MosA1 reference → WholeBrainA query","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-f23306b94dc7b6","kind":"model","name":"scVI","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["scxdr-2026"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scVI","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scVI is the method recorded for Cross-dataset single-cell drug response transfer. This page preserves the configuration reported by scXDR: drug response prediction across single-cell datasets via heterogeneous network transfer learning.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Single-cell-to-single-cell transfer; source scenario 2.","source_ids":["scxdr-2026"],"source_locator":"Table 2, scVI row, Scenario 2 column"}],"facts":[{"label":"Recorded dataset","value":"scXDR transfer scenario 2","source_ids":["scxdr-2026"],"source_locator":"Table 2, scVI row, Scenario 2 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["scxdr-2026"],"source_locator":"Table 2, scVI row, Scenario 2 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, scVI row, Scenario 2 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-f83c0b833411a7","kind":"model","name":"ESM-C","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["prime-2026"],"links":[],"attributes":{"entity_level":"method","version":"300M","reported_name":"ESM-C","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ESM-C is the method recorded for Mutated RBD binding prediction. This page preserves the configuration reported by PRIME: An evaluation framework for protein representation inference and generalization in viral mutation space.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Frozen mean-pooled representation with downstream regression; position-stratified split.","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"}],"facts":[{"label":"Recorded dataset","value":"PRIME mutated RBD","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"},{"label":"Recorded split","value":"position-stratified","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"},{"label":"Recorded configuration","value":"300M","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-f8f0257b98749a","kind":"model","name":"SPIN + ESM2-35M","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["spin-protein-function-2026"],"links":[],"attributes":{"entity_level":"method","version":"ESM2-35M frozen","reported_name":"SPIN + ESM2-35M","missing_metadata":{"checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"SPIN + ESM2-35M is the method recorded for protein function annotation. This page preserves the configuration reported by Scaling the profile of life by function with SPIN.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: frozen ESM2-35M backbone in SPIN","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"}],"facts":[{"label":"Recorded dataset","value":"TRX","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"},{"label":"Recorded split","value":"test set","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"},{"label":"Recorded configuration","value":"ESM2-35M frozen","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"}],"coverage":"limited","gaps":["The retained evidence location (Table 1, ESM2-35M Test row, F1_m-w column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-fa2da404b4d08e","kind":"model","name":"DEELIG","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["molecular-interactions"]},"source_ids":["deelig-2021"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"DEELIG","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"DEELIG is the method recorded for Protein–ligand binding affinity prediction. This page preserves the configuration reported by DEELIG: A Deep Learning Approach to Predict Protein-Ligand Binding Affinity.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Source paper reports DEELIG on PDBbind core set.","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column"}],"facts":[{"label":"Recorded dataset","value":"PDBbind core v2016","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, DEELIG row, PDBbind v2016 column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-fcf2cd29a81aae","kind":"model","name":"scGen","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["cells-tissues"]},"source_ids":["insilico-perturbation-auprc-2025"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"scGen","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"scGen is the method recorded for differentially expressed gene identification. This page preserves the configuration reported by AUPRC: a metric for evaluating the performance of in-silico perturbation methods in identifying differentially expressed genes.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: In-silico perturbation assessment with precision sampled at fixed 50% recall","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"}],"facts":[{"label":"Recorded dataset","value":"stimulated immune PBMC","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"},{"label":"Recorded split","value":"CD14+Mono","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"}],"coverage":"limited","gaps":["The retained evidence location (Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-model-fdac4c1ec8a433","kind":"model","name":"ProtT5 embeddings + ensemble classifier","description":"Identity as reported in this paper. Unspecified versions are not assumed equivalent to other papers.","status":"needs_review","facets":{"areas":["proteins-complexes"]},"source_ids":["protein-binding-sites-2023"],"links":[],"attributes":{"entity_level":"method","version":null,"reported_name":"ProtT5 embeddings + ensemble classifier","missing_metadata":{"version":"not_reported_in_legacy_extract","checkpoint_revision":"not_reported_in_legacy_extract","training_data":"not_reported_in_legacy_extract","licence":"not_reported_in_legacy_extract"},"profile":{"summary":"ProtT5 embeddings + ensemble classifier is the method recorded for protein-protein binding-site prediction. This page preserves the configuration reported by Learning the protein language of proteome-wide protein-protein binding sites via explainable ensemble deep learning.","sections":[{"title":"Recorded evaluation","body":"The imported evaluation describes this procedure: Explainable ensemble binding-site predictor using ProtT5 features","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"}],"facts":[{"label":"Recorded dataset","value":"Dset_448","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"}],"strengths":[],"limitations":[{"text":"This is an evidence-limited profile. The recorded evaluation context does not establish performance on other datasets or configurations.","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"}],"coverage":"limited","gaps":["The retained evidence location (Table 2, Dset_448 section, ProtT5 row, AUROC column) supports the reported evaluation, but this extraction does not establish the model's architecture, training corpus or checkpoint hash.","Unresolved registry fields: version (not reported in legacy extract), checkpoint revision (not reported in legacy extract), training data (not reported in legacy extract), licence (not reported in legacy extract)."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Reviewed the existing release record, its source pointer and linked evaluation context. This is not a fresh full-text architecture review or independent reproduction; numerical review status is unchanged."}}}} {"id":"reported-task-003d746a129c9b","kind":"benchmark","name":"differentially expressed gene identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["differentially expressed gene identification"]},"source_ids":["insilico-perturbation-auprc-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-7bf2cf7d2b2d01"}],"attributes":{"entity_level":"task","version":null,"task":"differentially expressed gene identification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests differentially expressed gene identification using stimulated immune PBMC.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: In-silico perturbation assessment with precision sampled at fixed 50% recall. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"},{"label":"Inputs","value":"stimulated immune PBMC","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"},{"label":"Assessment","value":"precision at 50% recall","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"},{"label":"Recorded split or evaluation setting","value":"CD14+Mono","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"}],"diagram":{"title":"Reported evaluation outline","steps":["stimulated immune PBMC","Recorded fitting or scoring procedure","Assess precision at 50% recall"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["insilico-perturbation-auprc-2025"],"source_locator":"Table 3, CD14+Mono section, scGen row, Precision at 50% Recall column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for stimulated immune PBMC have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-00e594df6a182d","kind":"benchmark","name":"Mutated RBD binding prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Mutated RBD binding prediction"]},"source_ids":["prime-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-becc215358afd0"}],"attributes":{"entity_level":"task","version":null,"task":"Mutated RBD binding prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Mutated RBD binding prediction using PRIME mutated RBD.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Frozen mean-pooled representation with downstream regression; position-stratified split. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column; Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column; Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"},{"label":"Inputs","value":"PRIME mutated RBD","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column; Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"},{"label":"Assessment","value":"R²","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column; Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"},{"label":"Recorded split or evaluation setting","value":"position-stratified","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column; Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column; Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column; Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"}],"diagram":{"title":"Reported evaluation outline","steps":["PRIME mutated RBD","Recorded fitting or scoring procedure","Assess R²"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["prime-2026"],"source_locator":"Table 1, ESM-2 8M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column; Table 1, ESM-C 300M / Mean / not fine-tuned row, Position-Stratified Split Binding R² column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for PRIME mutated RBD have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-016f70615f2cfc","kind":"benchmark","name":"RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["debfold-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-18ebde58579c2b"}],"attributes":{"entity_level":"task","version":null,"task":"RNA secondary structure","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests RNA secondary structure using DEBFold TestSetβ.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Median F1 on the prepared TestSetβ. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column; Table 1, RNAfold row, TestSetβ F1 (%) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column; Table 1, RNAfold row, TestSetβ F1 (%) column"},{"label":"Inputs","value":"DEBFold TestSetβ","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column; Table 1, RNAfold row, TestSetβ F1 (%) column"},{"label":"Assessment","value":"Median F1","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column; Table 1, RNAfold row, TestSetβ F1 (%) column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column; Table 1, RNAfold row, TestSetβ F1 (%) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column; Table 1, RNAfold row, TestSetβ F1 (%) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column; Table 1, RNAfold row, TestSetβ F1 (%) column"}],"diagram":{"title":"Reported evaluation outline","steps":["DEBFold TestSetβ","Recorded fitting or scoring procedure","Assess Median F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["debfold-2024"],"source_locator":"Table 1, DEBFold row, TestSetβ F1 (%) column; Table 1, RNAfold row, TestSetβ F1 (%) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for DEBFold TestSetβ have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-031186b57c62de","kind":"benchmark","name":"Human thymus cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Human thymus cell-type classification"]},"source_ids":["mouse-geneformer-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-477a9082515406"}],"attributes":{"entity_level":"task","version":null,"task":"Human thymus cell-type classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Human thymus cell-type classification using Human thymus scRNA-seq.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Ortholog-based gene conversion; zero-shot mouse model on human cells.; Native human model; zero-shot setting. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column; Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column; Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"},{"label":"Inputs","value":"Human thymus scRNA-seq","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column; Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"},{"label":"Assessment","value":"F1","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column; Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column; Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column; Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column; Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["Human thymus scRNA-seq","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["mouse-geneformer-2025"],"source_locator":"Table 4, h/ Thymus row, Mouse-Geneformer Zero-shot F1 column; Table 4, h/ Thymus row, Human-Geneformer Zero-shot F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Human thymus scRNA-seq have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-0647b0364def8f","kind":"benchmark","name":"antibody deamidation-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["antibody deamidation-site prediction"]},"source_ids":["antibody-deamidation-plm-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-0edd8f724db696"}],"attributes":{"entity_level":"task","version":null,"task":"antibody deamidation-site prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests antibody deamidation-site prediction using antibody peptide-mapping training dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: global contextual embeddings only. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"},{"label":"Inputs","value":"antibody peptide-mapping training dataset","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"},{"label":"Assessment","value":"accuracy","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"},{"label":"Recorded split or evaluation setting","value":"fivefold stratified CV","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"}],"diagram":{"title":"Reported evaluation outline","steps":["antibody peptide-mapping training dataset","Recorded fitting or scoring procedure","Assess accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["antibody-deamidation-plm-2024"],"source_locator":"Table 1, Global embeddings only row, Accuracy column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for antibody peptide-mapping training dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-09c3100b77dcc5","kind":"benchmark","name":"protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["protein-protein interaction prediction"]},"source_ids":["esm2-amp-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-9135087a16af1c"}],"attributes":{"entity_level":"task","version":null,"task":"protein-protein interaction prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests protein-protein interaction prediction using Bernett PPI dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: ESM2-derived embeddings plus paper interaction predictor. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"},{"label":"Inputs","value":"Bernett PPI dataset","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["Bernett PPI dataset","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["esm2-amp-2025"],"source_locator":"Table 4, ESM2_AMPS row, AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Bernett PPI dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-0c92cda11228c4","kind":"benchmark","name":"Antibody–antigen interaction prediction using folded complexes","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody–antigen interaction prediction using folded complexes"]},"source_ids":["antibody-flexibility-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-ee26acbd6e8cf7"}],"attributes":{"entity_level":"task","version":null,"task":"Antibody–antigen interaction prediction using folded complexes","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Antibody–antigen interaction prediction using folded complexes using Antibody–antigen GEP test set.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Interaction classifier evaluated using Chai-1-folded input complexes; this is pipeline AUC, not DockQ.; Interaction classifier evaluated using Boltz-1-folded input complexes; this is pipeline AUC, not DockQ. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column; Table 5, Folded row, Boltz-1 (no MSA) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column; Table 5, Folded row, Boltz-1 (no MSA) column"},{"label":"Inputs","value":"Antibody–antigen GEP test set","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column; Table 5, Folded row, Boltz-1 (no MSA) column"},{"label":"Assessment","value":"AUC-ROC","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column; Table 5, Folded row, Boltz-1 (no MSA) column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column; Table 5, Folded row, Boltz-1 (no MSA) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column; Table 5, Folded row, Boltz-1 (no MSA) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column; Table 5, Folded row, Boltz-1 (no MSA) column"}],"diagram":{"title":"Reported evaluation outline","steps":["Antibody–antigen GEP test set","Recorded fitting or scoring procedure","Assess AUC-ROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["antibody-flexibility-2025"],"source_locator":"Table 5, Folded row, Chai-1 (no MSA) column; Table 5, Folded row, Boltz-1 (no MSA) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Antibody–antigen GEP test set have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-132da895d4c381","kind":"benchmark","name":"Enhancer classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer classification"]},"source_ids":["enbed-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-f0bf60a62ad7c3"}],"attributes":{"entity_level":"task","version":null,"task":"Enhancer classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Enhancer classification using Genomic Benchmarks Mouse Enhancers.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Reported Genomic Benchmarks classification accuracy.; ENBED trained on GRCh38; reported Genomic Benchmarks classification accuracy. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column; Table 2, Mouse Enhancers row, ENBED (GRCh38) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column; Table 2, Mouse Enhancers row, ENBED (GRCh38) column"},{"label":"Inputs","value":"Genomic Benchmarks Mouse Enhancers","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column; Table 2, Mouse Enhancers row, ENBED (GRCh38) column"},{"label":"Assessment","value":"Accuracy","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column; Table 2, Mouse Enhancers row, ENBED (GRCh38) column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column; Table 2, Mouse Enhancers row, ENBED (GRCh38) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column; Table 2, Mouse Enhancers row, ENBED (GRCh38) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column; Table 2, Mouse Enhancers row, ENBED (GRCh38) column"}],"diagram":{"title":"Reported evaluation outline","steps":["Genomic Benchmarks Mouse Enhancers","Recorded fitting or scoring procedure","Assess Accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["enbed-2024"],"source_locator":"Table 2, Mouse Enhancers row, ENBED column; Table 2, Mouse Enhancers row, ENBED (GRCh38) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Genomic Benchmarks Mouse Enhancers have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-13dfe6b33e71ed","kind":"benchmark","name":"polyadenylation site detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["polyadenylation site detection"]},"source_ids":["polya-glm-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-55f200c9481409"}],"attributes":{"entity_level":"task","version":null,"task":"polyadenylation site detection","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests polyadenylation site detection using poly(A) Gene-Gene.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: few-shot Gene-Gene negative-set comparison. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"},{"label":"Inputs","value":"poly(A) Gene-Gene","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"},{"label":"Assessment","value":"AUC","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"},{"label":"Recorded split or evaluation setting","value":"5-fold cross-validation","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"}],"diagram":{"title":"Reported evaluation outline","steps":["poly(A) Gene-Gene","Recorded fitting or scoring procedure","Assess AUC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["polya-glm-2025"],"source_locator":"Table 1, Few-shot HyenaDNA row, G-G AUC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for poly(A) Gene-Gene have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-167f08013c270e","kind":"benchmark","name":"Cross-dataset single-cell drug response transfer","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-dataset single-cell drug response transfer"]},"source_ids":["scxdr-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-f7210686a78474"}],"attributes":{"entity_level":"task","version":null,"task":"Cross-dataset single-cell drug response transfer","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Cross-dataset single-cell drug response transfer using scXDR transfer scenario 2.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Single-cell-to-single-cell transfer; source scenario 2. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column; Table 2, scVI row, Scenario 2 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column; Table 2, scVI row, Scenario 2 column"},{"label":"Inputs","value":"scXDR transfer scenario 2","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column; Table 2, scVI row, Scenario 2 column"},{"label":"Assessment","value":"AUC","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column; Table 2, scVI row, Scenario 2 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column; Table 2, scVI row, Scenario 2 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column; Table 2, scVI row, Scenario 2 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column; Table 2, scVI row, Scenario 2 column"}],"diagram":{"title":"Reported evaluation outline","steps":["scXDR transfer scenario 2","Recorded fitting or scoring procedure","Assess AUC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["scxdr-2026"],"source_locator":"Table 2, scXDR row, Scenario 2 column; Table 2, scVI row, Scenario 2 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for scXDR transfer scenario 2 have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-1c74661df2c401","kind":"benchmark","name":"flu-vaccine mRNA property prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["flu-vaccine mRNA property prediction"]},"source_ids":["codonbert-vaccines-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-54b9bc432928d6"}],"attributes":{"entity_level":"task","version":null,"task":"flu-vaccine mRNA property prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests flu-vaccine mRNA property prediction using flu-vaccine sequences.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: codon-based model fine-tuned for downstream regression. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"},{"label":"Inputs","value":"flu-vaccine sequences","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"},{"label":"Assessment","value":"Spearman rho","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"},{"label":"Recorded split or evaluation setting","value":"paper evaluation","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"}],"diagram":{"title":"Reported evaluation outline","steps":["flu-vaccine sequences","Recorded fitting or scoring procedure","Assess Spearman rho"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["codonbert-vaccines-2024"],"source_locator":"Table 2, CodonBERT row, Flu vaccines Spearman correlation column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for flu-vaccine sequences have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-1ebf9b408517f9","kind":"benchmark","name":"Enzyme functional identity prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Enzyme functional identity prediction"]},"source_ids":["fujisan-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-5197cca532f89d"}],"attributes":{"entity_level":"task","version":null,"task":"Enzyme functional identity prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Enzyme functional identity prediction using FUJISAN test sub-dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Sequence and structural feature integration; paper-reported test sub-dataset.; Comparator evaluated on the paper-reported test sub-dataset. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column; Table 1, ESM2 row, AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column; Table 1, ESM2 row, AUROC column"},{"label":"Inputs","value":"FUJISAN test sub-dataset","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column; Table 1, ESM2 row, AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column; Table 1, ESM2 row, AUROC column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column; Table 1, ESM2 row, AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column; Table 1, ESM2 row, AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column; Table 1, ESM2 row, AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["FUJISAN test sub-dataset","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["fujisan-2024"],"source_locator":"Table 1, FUJISAN row, AUROC column; Table 1, ESM2 row, AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for FUJISAN test sub-dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-22024610c4d658","kind":"benchmark","name":"enhancer prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["hi-enhancer-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-a8610f2b80cdf0"}],"attributes":{"entity_level":"task","version":null,"task":"enhancer prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests enhancer prediction using enhancer independent comparison.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Two-stage Hi-Enhancer system; paper Table 2 method comparison. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"},{"label":"Inputs","value":"enhancer independent comparison","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"},{"label":"Assessment","value":"accuracy","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"}],"diagram":{"title":"Reported evaluation outline","steps":["enhancer independent comparison","Recorded fitting or scoring procedure","Assess accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["hi-enhancer-2025"],"source_locator":"Table 2, Ours (Stacking-Auto) row, Accuracy column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for enhancer independent comparison have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-2cbac97dd849f5","kind":"benchmark","name":"Enhancer-target gene prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Enhancer-target gene prediction"]},"source_ids":["dnalongbench-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-fcb5752d916d5d"}],"attributes":{"entity_level":"task","version":null,"task":"Enhancer-target gene prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Enhancer-target gene prediction using DNALongBench ETGP.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Long-range ETGP benchmark; source table reports AUROC. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column; Table 3, Caduceus-Ph row, ETGP column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column; Table 3, Caduceus-Ph row, ETGP column"},{"label":"Inputs","value":"DNALongBench ETGP","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column; Table 3, Caduceus-Ph row, ETGP column"},{"label":"Assessment","value":"AUROC","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column; Table 3, Caduceus-Ph row, ETGP column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column; Table 3, Caduceus-Ph row, ETGP column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column; Table 3, Caduceus-Ph row, ETGP column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column; Table 3, Caduceus-Ph row, ETGP column"}],"diagram":{"title":"Reported evaluation outline","steps":["DNALongBench ETGP","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["dnalongbench-2025"],"source_locator":"Table 3, HyenaDNA row, ETGP column; Table 3, Caduceus-Ph row, ETGP column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for DNALongBench ETGP have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-3063ed4da76b4b","kind":"benchmark","name":"Gene-regulatory link prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Gene-regulatory link prediction"]},"source_ids":["scregnet-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-2ad2fad5e1cd0a"}],"attributes":{"entity_level":"task","version":null,"task":"Gene-regulatory link prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Gene-regulatory link prediction using hESC cell-type-specific GRN.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: TFs plus 500 variable genes; mean from 50 independent evaluations. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry; Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry; Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"},{"label":"Inputs","value":"hESC cell-type-specific GRN","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry; Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"},{"label":"Assessment","value":"AUROC","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry; Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry; Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry; Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry; Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"}],"diagram":{"title":"Reported evaluation outline","steps":["hESC cell-type-specific GRN","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["scregnet-2025"],"source_locator":"Table 2, scRegNet (w/ Geneformer) row, hESC AUROC entry; Table 2, scRegNet (w/ scBERT) row, hESC AUROC entry"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for hESC cell-type-specific GRN have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-3109f8d0f2b7b5","kind":"benchmark","name":"cross-species conservation prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["cross-species conservation prediction"]},"source_ids":["plantcad2-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-b6d0ebaca196a6"}],"attributes":{"entity_level":"task","version":null,"task":"cross-species conservation prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests cross-species conservation prediction using Andropogoneae genome-wide conservation.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Zero-shot score for conserved versus non-conserved sites from alignments of 35 Andropogoneae genomes. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"},{"label":"Inputs","value":"Andropogoneae genome-wide conservation","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"},{"label":"Assessment","value":"AUROC","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"}],"diagram":{"title":"Reported evaluation outline","steps":["Andropogoneae genome-wide conservation","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["plantcad2-2025"],"source_locator":"Table 1, Cross-species evolutionary conservation > Conservation within Andropogoneae (Genome-wide) row, PlantCAD2 AUROC entry"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Andropogoneae genome-wide conservation have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-369dcfef14c4a9","kind":"benchmark","name":"Simulated metagenome virus-taxon retrieval","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated metagenome virus-taxon retrieval"]},"source_ids":["lazypipe-2020"],"links":[{"relation":"dataset","target_id":"reported-dataset-450c1af18cc623"}],"attributes":{"entity_level":"task","version":null,"task":"Simulated metagenome virus-taxon retrieval","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Simulated metagenome virus-taxon retrieval using Simulated viral metagenome.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Genus-rank viral taxon retrieval. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column; Table 1, Kraken2 / Genus row, F column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column; Table 1, Kraken2 / Genus row, F column"},{"label":"Inputs","value":"Simulated viral metagenome","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column; Table 1, Kraken2 / Genus row, F column"},{"label":"Assessment","value":"Genus-level F1","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column; Table 1, Kraken2 / Genus row, F column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column; Table 1, Kraken2 / Genus row, F column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column; Table 1, Kraken2 / Genus row, F column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column; Table 1, Kraken2 / Genus row, F column"}],"diagram":{"title":"Reported evaluation outline","steps":["Simulated viral metagenome","Recorded fitting or scoring procedure","Assess Genus-level F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["lazypipe-2020"],"source_locator":"Table 1, Lazypipe-nt / Genus row, F column; Table 1, Kraken2 / Genus row, F column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Simulated viral metagenome have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-3891811dcce8b3","kind":"benchmark","name":"E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["prokbert-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-48def1da574597"}],"attributes":{"entity_level":"task","version":null,"task":"E. coli sigma70 promoter prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests E. coli sigma70 promoter prediction using E. coli sigma70 promoter dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Promoter versus non-promoter classification. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column; Table 3, Promotech row, Accuracy column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column; Table 3, Promotech row, Accuracy column"},{"label":"Inputs","value":"E. coli sigma70 promoter dataset","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column; Table 3, Promotech row, Accuracy column"},{"label":"Assessment","value":"Accuracy","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column; Table 3, Promotech row, Accuracy column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column; Table 3, Promotech row, Accuracy column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column; Table 3, Promotech row, Accuracy column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column; Table 3, Promotech row, Accuracy column"}],"diagram":{"title":"Reported evaluation outline","steps":["E. coli sigma70 promoter dataset","Recorded fitting or scoring procedure","Assess Accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["prokbert-2024"],"source_locator":"Table 3, ProkBERT-mini row, Accuracy column; Table 3, Promotech row, Accuracy column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for E. coli sigma70 promoter dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-3a3bff34cce634","kind":"benchmark","name":"RNA compound-binding site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA compound-binding site prediction"]},"source_ids":["cobra-rna-binding-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-b1af840b76b351"}],"attributes":{"entity_level":"task","version":null,"task":"RNA compound-binding site prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests RNA compound-binding site prediction using CoBRA compound-binding test set.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: ERNIE-RNA embedding with TCL focal loss. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"},{"label":"Inputs","value":"CoBRA compound-binding test set","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"},{"label":"Assessment","value":"MCC","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"},{"label":"Recorded split or evaluation setting","value":"test set","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"}],"diagram":{"title":"Reported evaluation outline","steps":["CoBRA compound-binding test set","Recorded fitting or scoring procedure","Assess MCC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["cobra-rna-binding-2026"],"source_locator":"Table 2, ERNIE-RNA / TCL focal row, MCC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for CoBRA compound-binding test set have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-3d4dec23120fef","kind":"benchmark","name":"viral sequence detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["viral sequence detection"]},"source_ids":["detire-viral-metagenomes-2023"],"links":[{"relation":"dataset","target_id":"reported-dataset-0b54f42a987b1d"}],"attributes":{"entity_level":"task","version":null,"task":"viral sequence detection","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests viral sequence detection using testing viral metagenome dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Hybrid deep learning virus-fragment classifier on paper testing dataset. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"},{"label":"Inputs","value":"testing viral metagenome dataset","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"},{"label":"Assessment","value":"accuracy","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"},{"label":"Recorded split or evaluation setting","value":"test","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"}],"diagram":{"title":"Reported evaluation outline","steps":["testing viral metagenome dataset","Recorded fitting or scoring procedure","Assess accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["detire-viral-metagenomes-2023"],"source_locator":"Table 1, Accuracy row, DETIRE column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for testing viral metagenome dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-4420dcdfe8338d","kind":"benchmark","name":"metagenomic genus classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["metagenomic genus classification"]},"source_ids":["pc-mer-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-d28955d5872903"}],"attributes":{"entity_level":"task","version":null,"task":"metagenomic genus classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests metagenomic genus classification using AMP.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: k=8 PC-mer feature extraction with logistic regression on AMP genus-classification dataset. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"},{"label":"Inputs","value":"AMP","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"},{"label":"Assessment","value":"accuracy","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"},{"label":"Recorded split or evaluation setting","value":"genus-level","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"}],"diagram":{"title":"Reported evaluation outline","steps":["AMP","Recorded fitting or scoring procedure","Assess accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["pc-mer-2024"],"source_locator":"Table 3, AMP section, PC-mer + LR k=8 row, Accuracy (%) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for AMP have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-45105e1c486251","kind":"benchmark","name":"CAMI II phylum read classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["CAMI II phylum read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-beb4f5da29da0a"}],"attributes":{"entity_level":"task","version":null,"task":"CAMI II phylum read classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests CAMI II phylum read classification using CAMI II Sample_0 10,000-read subsample.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Phylum-level macro-averaged F1; distinct taxonomic rank from the other row. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Phylum row, F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Phylum row, F1 column"},{"label":"Inputs","value":"CAMI II Sample_0 10,000-read subsample","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Phylum row, F1 column"},{"label":"Assessment","value":"Macro F1","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Phylum row, F1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Phylum row, F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Phylum row, F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Phylum row, F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["CAMI II Sample_0 10,000-read subsample","Recorded fitting or scoring procedure","Assess Macro F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Phylum row, F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for CAMI II Sample_0 10,000-read subsample have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-45ead9a1eddf8d","kind":"benchmark","name":"antigen-antibody HIV neutralization prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["antigen-antibody HIV neutralization prediction"]},"source_ids":["deepinteraware-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-50f0bdb7cf9ca4"}],"attributes":{"entity_level":"task","version":null,"task":"antigen-antibody HIV neutralization prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests antigen-antibody HIV neutralization prediction using HIV neutralization.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Sequence-based interface-aware model, antibody-unseen split. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"},{"label":"Inputs","value":"HIV neutralization","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"},{"label":"Recorded split or evaluation setting","value":"antibody-unseen","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["HIV neutralization","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["deepinteraware-2025"],"source_locator":"Table 2, Ab Unseen section, DeepInterAware row, AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for HIV neutralization have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-46e927bea10702","kind":"benchmark","name":"miRNA-mRNA interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["miRNA-mRNA interaction prediction"]},"source_ids":["rnaret-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-99afd0c86b2954"}],"attributes":{"entity_level":"task","version":null,"task":"miRNA-mRNA interaction prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests miRNA-mRNA interaction prediction using MirTarRAW.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: 5-mer RNAret classifier; 72/8/20 train/validation/test split. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"},{"label":"Inputs","value":"MirTarRAW","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"},{"label":"Assessment","value":"F1","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"},{"label":"Recorded split or evaluation setting","value":"held-out test","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["MirTarRAW","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["rnaret-2026"],"source_locator":"Table 1, MirTarRAW / 5-mer RNAret row, F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for MirTarRAW have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-47465954d606e6","kind":"benchmark","name":"vaccine-antigen candidate prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["vaccine-antigen candidate prediction"]},"source_ids":["vaxign-esm-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-b91c871eb7740a"}],"attributes":{"entity_level":"task","version":null,"task":"vaccine-antigen candidate prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests vaccine-antigen candidate prediction using vaccine candidate validation.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Combined skip architecture, four layers, ESM-generated sequence features. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"},{"label":"Inputs","value":"vaccine candidate validation","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"},{"label":"Assessment","value":"AUPRC","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"}],"diagram":{"title":"Reported evaluation outline","steps":["vaccine candidate validation","Recorded fitting or scoring procedure","Assess AUPRC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["vaxign-esm-2024"],"source_locator":"Table 2, 4 Layers row, AUPRC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for vaccine candidate validation have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-4a54ce01b5a855","kind":"benchmark","name":"unseen-species genus classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["unseen-species genus classification"]},"source_ids":["barcodebert-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-bc127dc9c441fe"}],"attributes":{"entity_level":"task","version":null,"task":"unseen-species genus classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"Classify DNA barcodes from unseen species at the genus level.","sections":[{"title":"Procedure","body":"Use the training subset of the Seen partition as the reference and the Unseen partition as queries. Compare frozen sequence representations with cosine similarity and assign the label of the single nearest reference. Species are absent from other partitions, while their genera occur in the reference.","source_ids":["barcodebert-2026"],"source_locator":"BarcodeBERT paper: 3.1 Dataset / Data partitioning; genus-level 1-NN probing paragraph; Table 1"}],"facts":[{"label":"Record type","value":"Paper-specific nearest-neighbour protocol","source_ids":["barcodebert-2026"],"source_locator":"BarcodeBERT paper: 3.1 Dataset / Data partitioning; genus-level 1-NN probing paragraph; Table 1"},{"label":"Inputs","value":"DNA barcodes from the Canadian invertebrate reference library","source_ids":["barcodebert-2026"],"source_locator":"BarcodeBERT paper: 3.1 Dataset / Data partitioning; genus-level 1-NN probing paragraph; Table 1"},{"label":"Assessment","value":"Genus-level classification accuracy","source_ids":["barcodebert-2026"],"source_locator":"BarcodeBERT paper: 3.1 Dataset / Data partitioning; genus-level 1-NN probing paragraph; Table 1"},{"label":"Reported configuration","value":"BarcodeBERT (4–4–4); genus-level 1-NN probe, distinct from fine-tuned species classification","source_ids":["barcodebert-2026"],"source_locator":"Table 1, BarcodeBERT (4–4–4) row; 1-NN probe column"}],"strengths":[{"text":"A species-disjoint query set tests generalization within known genera.","source_ids":["barcodebert-2026"],"source_locator":"BarcodeBERT paper: 3.1 Dataset / Data partitioning; genus-level 1-NN probing paragraph; Table 1"}],"limitations":[{"text":"This is not unseen-genus classification or universal metagenomic recognition. BLAST is a relevant alignment comparator in the same paper.","source_ids":["barcodebert-2026"],"source_locator":"BarcodeBERT paper: 3.1 Dataset / Data partitioning; genus-level 1-NN probing paragraph; Table 1"}],"diagram":{"title":"Procedure overview","steps":["Seen reference barcodes","Embed unseen-species queries","Cosine nearest neighbour","Score genus labels"],"caption":"Conceptual overview, not an executable specification.","source_ids":["barcodebert-2026"],"source_locator":"BarcodeBERT paper: 3.1 Dataset / Data partitioning; genus-level 1-NN probing paragraph; Table 1"},"coverage":"reviewed","gaps":["The downloadable dataset revision and complete split manifest are not yet pinned in this profile."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}}} {"id":"reported-task-4df1fb456d3deb","kind":"benchmark","name":"Cell-type structure in frozen embeddings","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type structure in frozen embeddings"]},"source_ids":["genept-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-eaa2965545c87b"}],"attributes":{"entity_level":"task","version":null,"task":"Cell-type structure in frozen embeddings","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Cell-type structure in frozen embeddings using Aorta single-cell dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: k-means on pretrained cell embeddings; agreement with original cell-type labels. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column; Table 2, Aorta / Cell type row, scGPT ARI column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column; Table 2, Aorta / Cell type row, scGPT ARI column"},{"label":"Inputs","value":"Aorta single-cell dataset","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column; Table 2, Aorta / Cell type row, scGPT ARI column"},{"label":"Assessment","value":"Adjusted Rand Index","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column; Table 2, Aorta / Cell type row, scGPT ARI column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column; Table 2, Aorta / Cell type row, scGPT ARI column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column; Table 2, Aorta / Cell type row, scGPT ARI column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column; Table 2, Aorta / Cell type row, scGPT ARI column"}],"diagram":{"title":"Reported evaluation outline","steps":["Aorta single-cell dataset","Recorded fitting or scoring procedure","Assess Adjusted Rand Index"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["genept-2024"],"source_locator":"Table 2, Aorta / Cell type row, GenePT-w ARI column; Table 2, Aorta / Cell type row, scGPT ARI column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Aorta single-cell dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-53506fe386e4a1","kind":"benchmark","name":"human-versus-viral protein classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["human-versus-viral protein classification"]},"source_ids":["viral-immune-mimicry-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-43f24c4dfb7351"}],"attributes":{"entity_level":"task","version":null,"task":"human-versus-viral protein classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests human-versus-viral protein classification using human and viral proteins.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: ESM2 650M embedding-based human-virus classifier. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"},{"label":"Inputs","value":"human and viral proteins","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"},{"label":"Assessment","value":"AUROC","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"}],"diagram":{"title":"Reported evaluation outline","steps":["human and viral proteins","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["viral-immune-mimicry-2025"],"source_locator":"Table 1, ESM2 650M row, AUC (%) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for human and viral proteins have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-53e3d216eef6db","kind":"benchmark","name":"Simulated prophage-contig detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Simulated prophage-contig detection"]},"source_ids":["viral-contig-simulation-2021"],"links":[{"relation":"dataset","target_id":"reported-dataset-cd51026cdb6a7a"}],"attributes":{"entity_level":"task","version":null,"task":"Simulated prophage-contig detection","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Simulated prophage-contig detection using 20 medium/high-complexity viral simulations.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Average across twenty medium- and high-complexity simulated communities. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column; Table 3, VirSorter row, Prophage F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column; Table 3, VirSorter row, Prophage F1 column"},{"label":"Inputs","value":"20 medium/high-complexity viral simulations","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column; Table 3, VirSorter row, Prophage F1 column"},{"label":"Assessment","value":"Average prophage F1","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column; Table 3, VirSorter row, Prophage F1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column; Table 3, VirSorter row, Prophage F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column; Table 3, VirSorter row, Prophage F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column; Table 3, VirSorter row, Prophage F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["20 medium/high-complexity viral simulations","Recorded fitting or scoring procedure","Assess Average prophage F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["viral-contig-simulation-2021"],"source_locator":"Table 3, Vibrant row, Prophage F1 column; Table 3, VirSorter row, Prophage F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for 20 medium/high-complexity viral simulations have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-5693847493f19f","kind":"benchmark","name":"mRNA half-life prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA half-life prediction"]},"source_ids":["mrna-lm-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-52f00ccaabf0d9"}],"attributes":{"entity_level":"task","version":null,"task":"mRNA half-life prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests mRNA half-life prediction using mRNA half-life.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: average test performance across cross-validation splits. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"},{"label":"Inputs","value":"mRNA half-life","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"},{"label":"Assessment","value":"Spearman rho","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"},{"label":"Recorded split or evaluation setting","value":"test set across CV splits","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"}],"diagram":{"title":"Reported evaluation outline","steps":["mRNA half-life","Recorded fitting or scoring procedure","Assess Spearman rho"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["mrna-lm-2025"],"source_locator":"Table 1, mRNA-LM row, mRNA half-life Spearman column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for mRNA half-life have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-571f0a2e7faed3","kind":"benchmark","name":"Strain-level abundance quantification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Strain-level abundance quantification"]},"source_ids":["cammiq-2022"],"links":[{"relation":"dataset","target_id":"reported-dataset-72e837e5b97041"}],"attributes":{"entity_level":"task","version":null,"task":"Strain-level abundance quantification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Strain-level abundance quantification using HumanGut-all strain-level query.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Strain-level quantification on the HumanGut-all synthetic query. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column; Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column; Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"},{"label":"Inputs","value":"HumanGut-all strain-level query","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column; Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"},{"label":"Assessment","value":"L1 abundance error","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column; Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column; Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column; Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column; Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"}],"diagram":{"title":"Reported evaluation outline","steps":["HumanGut-all strain-level query","Recorded fitting or scoring procedure","Assess L1 abundance error"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["cammiq-2022"],"source_locator":"Table 5, B. L1 Err. / HumanGut-all row, CAMMiQ L1 Err. column; Table 5, B. L1 Err. / HumanGut-all row, Kraken2 L1 Err. column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for HumanGut-all strain-level query have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-57dc3dcdb67a81","kind":"benchmark","name":"Mean ribosome load from MPRA","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["Mean ribosome load from MPRA"]},"source_ids":["mrnabench-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-3a3e3880a3fed0"}],"attributes":{"entity_level":"task","version":null,"task":"Mean ribosome load from MPRA","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Mean ribosome load from MPRA using mRNABench MRL-MPRA.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Linear probe; mean across ten random seeds. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column; Table 2, RNA-FM row, MRL MPRA column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column; Table 2, RNA-FM row, MRL MPRA column"},{"label":"Inputs","value":"mRNABench MRL-MPRA","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column; Table 2, RNA-FM row, MRL MPRA column"},{"label":"Assessment","value":"Pearson R","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column; Table 2, RNA-FM row, MRL MPRA column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column; Table 2, RNA-FM row, MRL MPRA column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column; Table 2, RNA-FM row, MRL MPRA column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column; Table 2, RNA-FM row, MRL MPRA column"}],"diagram":{"title":"Reported evaluation outline","steps":["mRNABench MRL-MPRA","Recorded fitting or scoring procedure","Assess Pearson R"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["mrnabench-2025"],"source_locator":"Table 2, RiNALMo row, MRL MPRA column; Table 2, RNA-FM row, MRL MPRA column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for mRNABench MRL-MPRA have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-5b929593eefc76","kind":"benchmark","name":"Cell-type identification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type identification"]},"source_ids":["single-cell-peft-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-488d5de6bb9c1b"}],"attributes":{"entity_level":"task","version":null,"task":"Cell-type identification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Cell-type identification using M.S. single-cell dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Native scLLM cell-type identification as reported in Table 2. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column; Table 2, M.S. / Geneformer row, F1-Score column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column; Table 2, M.S. / Geneformer row, F1-Score column"},{"label":"Inputs","value":"M.S. single-cell dataset","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column; Table 2, M.S. / Geneformer row, F1-Score column"},{"label":"Assessment","value":"F1-Score","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column; Table 2, M.S. / Geneformer row, F1-Score column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column; Table 2, M.S. / Geneformer row, F1-Score column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column; Table 2, M.S. / Geneformer row, F1-Score column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column; Table 2, M.S. / Geneformer row, F1-Score column"}],"diagram":{"title":"Reported evaluation outline","steps":["M.S. single-cell dataset","Recorded fitting or scoring procedure","Assess F1-Score"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["single-cell-peft-2024"],"source_locator":"Table 2, M.S. / scGPT row, F1-Score column; Table 2, M.S. / Geneformer row, F1-Score column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for M.S. single-cell dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-5ec7581b246ea6","kind":"benchmark","name":"RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["tu-fold-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-f2e729f333a333"}],"attributes":{"entity_level":"task","version":null,"task":"RNA secondary structure","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests RNA secondary structure using RNA8F.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Three-fold training and evaluation; source reports mean and standard deviation. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column; Table 2, UFold row, Overall F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column; Table 2, UFold row, Overall F1 column"},{"label":"Inputs","value":"RNA8F","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column; Table 2, UFold row, Overall F1 column"},{"label":"Assessment","value":"F1","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column; Table 2, UFold row, Overall F1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column; Table 2, UFold row, Overall F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column; Table 2, UFold row, Overall F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column; Table 2, UFold row, Overall F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["RNA8F","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["tu-fold-2025"],"source_locator":"Table 2, TU-Fold (aug) row, Overall F1 column; Table 2, UFold row, Overall F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for RNA8F have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-6243658a1bc215","kind":"benchmark","name":"Zero-shot substitution mutation effects: stability","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot substitution mutation effects: stability"]},"source_ids":["proteingym-2023"],"links":[{"relation":"dataset","target_id":"reported-dataset-7cec655cd742f3"}],"attributes":{"entity_level":"task","version":null,"task":"Zero-shot substitution mutation effects: stability","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Zero-shot substitution mutation effects: stability using ProteinGym substitution DMS: stability assays.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Zero-shot mutation scores; average Spearman across stability-category assays. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column; Table A7, ProteinMPNN row, Stability column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column; Table A7, ProteinMPNN row, Stability column"},{"label":"Inputs","value":"ProteinGym substitution DMS: stability assays","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column; Table A7, ProteinMPNN row, Stability column"},{"label":"Assessment","value":"Mean Spearman rho","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column; Table A7, ProteinMPNN row, Stability column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column; Table A7, ProteinMPNN row, Stability column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column; Table A7, ProteinMPNN row, Stability column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column; Table A7, ProteinMPNN row, Stability column"}],"diagram":{"title":"Reported evaluation outline","steps":["ProteinGym substitution DMS: stability assays","Recorded fitting or scoring procedure","Assess Mean Spearman rho"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["proteingym-2023"],"source_locator":"Table A7, ESM-2 (15B) row, Stability column; Table A7, ProteinMPNN row, Stability column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for ProteinGym substitution DMS: stability assays have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-6312c8a7ac045e","kind":"benchmark","name":"cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["cell-type annotation"]},"source_ids":["gremln-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-59def895fbdbb4"}],"attributes":{"entity_level":"task","version":null,"task":"cell-type annotation","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests cell-type annotation using non-immune cells.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Zero-shot cell-type annotation using pre-trained cellular graph foundation model. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"},{"label":"Inputs","value":"non-immune cells","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"},{"label":"Assessment","value":"F1","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"},{"label":"Recorded split or evaluation setting","value":"zero-shot","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"}],"diagram":{"title":"Reported evaluation outline","steps":["non-immune cells","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["gremln-2026"],"source_locator":"Table 2, Cell type annotation(zero-shot), Non-immune cells, F1 row, GREmLN column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for non-immune cells have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-6330d593980b5b","kind":"benchmark","name":"Long-read taxonomic profiling","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Long-read taxonomic profiling"]},"source_ids":["lemur-magnet-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-768a7ff5bac414"}],"attributes":{"entity_level":"task","version":null,"task":"Long-read taxonomic profiling","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Long-read taxonomic profiling using Zymo LOG 10%.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Mean across five replicate runs on Zymo LOG 10%. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column; Table 3, LOG 10% / Kraken 2 row, F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column; Table 3, LOG 10% / Kraken 2 row, F1 column"},{"label":"Inputs","value":"Zymo LOG 10%","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column; Table 3, LOG 10% / Kraken 2 row, F1 column"},{"label":"Assessment","value":"F1","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column; Table 3, LOG 10% / Kraken 2 row, F1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column; Table 3, LOG 10% / Kraken 2 row, F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column; Table 3, LOG 10% / Kraken 2 row, F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column; Table 3, LOG 10% / Kraken 2 row, F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["Zymo LOG 10%","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["lemur-magnet-2024"],"source_locator":"Table 3, LOG 10% / Lemur row, F1 column; Table 3, LOG 10% / Kraken 2 row, F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Zymo LOG 10% have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-64607443a9ba15","kind":"benchmark","name":"enhancer prediction","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer prediction"]},"source_ids":["enhancer-position-encoding-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-a03b8e9efde37b"}],"attributes":{"entity_level":"task","version":null,"task":"enhancer prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests enhancer prediction using human enhancer dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Nucleotide position-aware feature encoding; average assessment of CNN classifier. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"},{"label":"Inputs","value":"human enhancer dataset","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"},{"label":"Assessment","value":"AUROC","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"}],"diagram":{"title":"Reported evaluation outline","steps":["human enhancer dataset","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["enhancer-position-encoding-2024"],"source_locator":"Table 2, Human section, CNN row, AUC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for human enhancer dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-660753ec94e631","kind":"benchmark","name":"Cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cell-type annotation"]},"source_ids":["scelmo-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-8f123f006964ad"}],"attributes":{"entity_level":"task","version":null,"task":"Cell-type annotation","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Cell-type annotation using hPancreas.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Zero-shot setting; source caption says some comparator rows come from GenePT. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column; Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column; Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"},{"label":"Inputs","value":"hPancreas","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column; Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"},{"label":"Assessment","value":"F1","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column; Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column; Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column; Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column; Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"},{"text":"Some comparator provenance is quoted or unresolved. Do not treat copied comparisons as independent evaluations or assume identical protocols.","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column; Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["hPancreas","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["scelmo-2025"],"source_locator":"Table 1, hPancreas zero-shot / scGPT (z) row, F1 column; Table 1, hPancreas zero-shot / Geneformer (z) row, F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for hPancreas have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-6e54c7452b2b81","kind":"benchmark","name":"human protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["human protein-protein interaction prediction"]},"source_ids":["mulan-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-38151fa548e291"}],"attributes":{"entity_level":"task","version":null,"task":"human protein-protein interaction prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests human protein-protein interaction prediction using HumanPPI.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: MULAN sequence-structure model based on ESM2 8M. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"},{"label":"Inputs","value":"HumanPPI","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"},{"label":"Assessment","value":"AUC","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"},{"label":"Recorded split or evaluation setting","value":"paper evaluation","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"}],"diagram":{"title":"Reported evaluation outline","steps":["HumanPPI","Recorded fitting or scoring procedure","Assess AUC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["mulan-2025"],"source_locator":"Table 2, MULAN-ESM2 S row, HumanPPI AUC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for HumanPPI have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-7621fa1be55362","kind":"benchmark","name":"protein localization classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["protein localization classification"]},"source_ids":["cell-dino-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-d356eac961cb69"}],"attributes":{"entity_level":"task","version":null,"task":"protein localization classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests protein localization classification using HPA-FoV.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Self-supervised microscopy embedding pre-trained on HPA-FoV; downstream protein-localization classifier. Dataset-specific pretraining; the paper does not claim a general-purpose foundation model that generalizes beyond these benchmarks. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"},{"label":"Inputs","value":"HPA-FoV","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"},{"label":"Assessment","value":"F1","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"}],"diagram":{"title":"Reported evaluation outline","steps":["HPA-FoV","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["cell-dino-2025"],"source_locator":"Table 2, HPA-FoV section, Cell-DINO row, PL column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for HPA-FoV have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-77a32496ce8fe6","kind":"benchmark","name":"protein-ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["protein-ligand binding affinity prediction"]},"source_ids":["ligand-affinity-meta-model-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-17132fbabd7683"}],"attributes":{"entity_level":"task","version":null,"task":"protein-ligand binding affinity prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests protein-ligand binding affinity prediction using CASF-2016.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Sequence-or-structure meta-model; predicts ln(Kd/Ki) using docked and deep-learning components. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"},{"label":"Inputs","value":"CASF-2016","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"},{"label":"Assessment","value":"Pearson correlation","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"},{"label":"Recorded split or evaluation setting","value":"core benchmark","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"}],"diagram":{"title":"Reported evaluation outline","steps":["CASF-2016","Recorded fitting or scoring procedure","Assess Pearson correlation"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["ligand-affinity-meta-model-2024"],"source_locator":"Table 4, Meta-models row, CASF-2016 Benchmark > PCC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for CASF-2016 have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-786c09824e9bf5","kind":"benchmark","name":"clathrin protein classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["clathrin protein classification"]},"source_ids":["clathrin-plm-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-0aab382ca2c063"}],"attributes":{"entity_level":"task","version":null,"task":"clathrin protein classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests clathrin protein classification using CLA-IND0.6.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: single-feature ESM-2 embedding comparison. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"},{"label":"Inputs","value":"CLA-IND0.6","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"},{"label":"Assessment","value":"accuracy","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"},{"label":"Recorded split or evaluation setting","value":"independent test","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"}],"diagram":{"title":"Reported evaluation outline","steps":["CLA-IND0.6","Recorded fitting or scoring procedure","Assess accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["clathrin-plm-2025"],"source_locator":"Table 2, Independent test / ESM-2 row, ACC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for CLA-IND0.6 have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-7efe245cc94ee5","kind":"benchmark","name":"Combinatorial cell-label classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Combinatorial cell-label classification"]},"source_ids":["cell2sentence-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-87e91d9d6e6f4b"}],"attributes":{"entity_level":"task","version":null,"task":"Combinatorial cell-label classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Combinatorial cell-label classification using L1000.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Partial-credit labels including cell type, perturbation, and dose. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column; Table 3, Partial label / Geneformer row, L1000 Acc column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column; Table 3, Partial label / Geneformer row, L1000 Acc column"},{"label":"Inputs","value":"L1000","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column; Table 3, Partial label / Geneformer row, L1000 Acc column"},{"label":"Assessment","value":"Partial-label accuracy","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column; Table 3, Partial label / Geneformer row, L1000 Acc column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column; Table 3, Partial label / Geneformer row, L1000 Acc column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column; Table 3, Partial label / Geneformer row, L1000 Acc column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column; Table 3, Partial label / Geneformer row, L1000 Acc column"}],"diagram":{"title":"Reported evaluation outline","steps":["L1000","Recorded fitting or scoring procedure","Assess Partial-label accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["cell2sentence-2024"],"source_locator":"Table 3, Partial label / C2S (GPT-2 Large) row, L1000 Acc column; Table 3, Partial label / Geneformer row, L1000 Acc column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for L1000 have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-82fc7843f07324","kind":"benchmark","name":"human RNA 2-prime-O-methylation site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["human RNA 2-prime-O-methylation site prediction"]},"source_ids":["2ome-lm-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-bd3d8e7d6cd196"}],"attributes":{"entity_level":"task","version":null,"task":"human RNA 2-prime-O-methylation site prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests human RNA 2-prime-O-methylation site prediction using human RNA 2OMe sites.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: pretrained RNA language model predictor. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"},{"label":"Inputs","value":"human RNA 2OMe sites","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"},{"label":"Assessment","value":"AUC","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"},{"label":"Recorded split or evaluation setting","value":"5-fold cross-validation","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"}],"diagram":{"title":"Reported evaluation outline","steps":["human RNA 2OMe sites","Recorded fitting or scoring procedure","Assess AUC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["2ome-lm-2025"],"source_locator":"Table 1, 2OMe-LM row, AUC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for human RNA 2OMe sites have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-83be0998084c91","kind":"benchmark","name":"protein variant-effect classification","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein variant-effect classification"]},"source_ids":["structure-informed-plm-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-2eaa2a051d45ee"}],"attributes":{"entity_level":"task","version":null,"task":"protein variant-effect classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests protein variant-effect classification using variant-effects benchmark.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: combined amino-acid, secondary structure, solvent accessibility and contact-map scoring. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"},{"label":"Inputs","value":"variant-effects benchmark","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"},{"label":"Recorded split or evaluation setting","value":"paper evaluation","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["variant-effects benchmark","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["structure-informed-plm-2025"],"source_locator":"PMC12068927 HTML, Table4, AA+SS+RSA+CM row, AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for variant-effects benchmark have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-8406b6aabfb8c0","kind":"benchmark","name":"Mock-community MAG taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Mock-community MAG taxonomy classification"]},"source_ids":["kmetashot-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-8cad416ddc80dc"}],"attributes":{"entity_level":"task","version":null,"task":"Mock-community MAG taxonomy classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Mock-community MAG taxonomy classification using Real mock community MAGs.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Genus classification of MAGs from MegaHIT contigs; uncorrected kMetaShot.; Genus classification of the same MAG set. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column; Table 2, F1-score % row, Genus Gtk column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column; Table 2, F1-score % row, Genus Gtk column"},{"label":"Inputs","value":"Real mock community MAGs","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column; Table 2, F1-score % row, Genus Gtk column"},{"label":"Assessment","value":"Genus-level F1","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column; Table 2, F1-score % row, Genus Gtk column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column; Table 2, F1-score % row, Genus Gtk column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column; Table 2, F1-score % row, Genus Gtk column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column; Table 2, F1-score % row, Genus Gtk column"}],"diagram":{"title":"Reported evaluation outline","steps":["Real mock community MAGs","Recorded fitting or scoring procedure","Assess Genus-level F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["kmetashot-2025"],"source_locator":"Table 2, F1-score % row, Genus kMS column; Table 2, F1-score % row, Genus Gtk column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Real mock community MAGs have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-86a628af87ff8f","kind":"benchmark","name":"enhancer recognition","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["enhancer recognition"]},"source_ids":["dnabert2-enhancer-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-6212e779949708"}],"attributes":{"entity_level":"task","version":null,"task":"enhancer recognition","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests enhancer recognition using Liu training dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: first-layer enhancer versus non-enhancer classifier. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"},{"label":"Inputs","value":"Liu training dataset","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"},{"label":"Assessment","value":"AUC","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"},{"label":"Recorded split or evaluation setting","value":"5-fold cross-validation","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"}],"diagram":{"title":"Reported evaluation outline","steps":["Liu training dataset","Recorded fitting or scoring procedure","Assess AUC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["dnabert2-enhancer-2025"],"source_locator":"Table 4, first-layer DNABERT2-Enhancer row, AUC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Liu training dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-92137759a9e7b0","kind":"benchmark","name":"Metagenomic taxonomic classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Metagenomic taxonomic classification"]},"source_ids":["nabas-plus-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-b462aa24561fba"}],"attributes":{"entity_level":"task","version":null,"task":"Metagenomic taxonomic classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Metagenomic taxonomic classification using CAMI II Toy human gastrooral sample19-new.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Newly generated sample19 used for classifier comparison. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column; Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column; Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"},{"label":"Inputs","value":"CAMI II Toy human gastrooral sample19-new","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column; Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"},{"label":"Assessment","value":"F1 score","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column; Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column; Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column; Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column; Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"}],"diagram":{"title":"Reported evaluation outline","steps":["CAMI II Toy human gastrooral sample19-new","Recorded fitting or scoring procedure","Assess F1 score"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["nabas-plus-2025"],"source_locator":"Table 3, Sample19-new / NABAS+ row, F1 score column; Table 3, Sample19-new / MetaPhlAn3 row, F1 score column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for CAMI II Toy human gastrooral sample19-new have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-94802534b7026d","kind":"benchmark","name":"Protein–ligand binding energy prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding energy prediction"]},"source_ids":["fingerprint-scoring-2022"],"links":[{"relation":"dataset","target_id":"reported-dataset-16d01b5ef88e84"}],"attributes":{"entity_level":"task","version":null,"task":"Protein–ligand binding energy prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Protein–ligand binding energy prediction using Fingerprint-scoring benchmark.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Binding-energy model using ligand and protein fingerprints with LightGBM.; PMF-only LASSO baseline evaluated by the same authors. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column; Table 1, PMF / LASSO row, R column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column; Table 1, PMF / LASSO row, R column"},{"label":"Inputs","value":"Fingerprint-scoring benchmark","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column; Table 1, PMF / LASSO row, R column"},{"label":"Assessment","value":"Pearson R","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column; Table 1, PMF / LASSO row, R column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column; Table 1, PMF / LASSO row, R column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column; Table 1, PMF / LASSO row, R column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column; Table 1, PMF / LASSO row, R column"}],"diagram":{"title":"Reported evaluation outline","steps":["Fingerprint-scoring benchmark","Recorded fitting or scoring procedure","Assess Pearson R"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["fingerprint-scoring-2022"],"source_locator":"Table 1, PMF + ECFP + PF / LightGBM row, R column; Table 1, PMF / LASSO row, R column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Fingerprint-scoring benchmark have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-988ff78f86471e","kind":"benchmark","name":"Human 5mC detection","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["Human 5mC detection"]},"source_ids":["dna-foundation-models-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-463197d6a98b99"}],"attributes":{"entity_level":"task","version":null,"task":"Human 5mC detection","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Human 5mC detection using Human 5mC.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Binary epigenetic-modification classification as reported in the paper. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column; Table 3, Human 5mC row, NT-v2 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column; Table 3, Human 5mC row, NT-v2 column"},{"label":"Inputs","value":"Human 5mC","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column; Table 3, Human 5mC row, NT-v2 column"},{"label":"Assessment","value":"AUC","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column; Table 3, Human 5mC row, NT-v2 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column; Table 3, Human 5mC row, NT-v2 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column; Table 3, Human 5mC row, NT-v2 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column; Table 3, Human 5mC row, NT-v2 column"}],"diagram":{"title":"Reported evaluation outline","steps":["Human 5mC","Recorded fitting or scoring procedure","Assess AUC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["dna-foundation-models-2025"],"source_locator":"Table 3, Human 5mC row, Caduceus-Ph column; Table 3, Human 5mC row, NT-v2 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Human 5mC have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-9917a0e69f33e7","kind":"benchmark","name":"DNA-binding residue prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["DNA-binding residue prediction"]},"source_ids":["megsite-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-739aee3cf8d6f1"}],"attributes":{"entity_level":"task","version":null,"task":"DNA-binding residue prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests DNA-binding residue prediction using DNA-129_Test.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: ESM3 multimodal embedding ablation in MegSite. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"},{"label":"Inputs","value":"DNA-129_Test","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"},{"label":"Assessment","value":"AUC","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"},{"label":"Recorded split or evaluation setting","value":"independent test","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"}],"diagram":{"title":"Reported evaluation outline","steps":["DNA-129_Test","Recorded fitting or scoring procedure","Assess AUC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["megsite-2025"],"source_locator":"Table 2, DNA-129_Test / ESM3 row, AUC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for DNA-129_Test have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-99afd88cb12895","kind":"benchmark","name":"gene-regulatory signal prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["gene-regulatory signal prediction"]},"source_ids":["single-cell-residual-geometry-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-d9fdd8dc7a0184"}],"attributes":{"entity_level":"task","version":null,"task":"gene-regulatory signal prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests gene-regulatory signal prediction using immune tissue.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Asymmetric extraction, PCA-64 centered cosine geometry added to scGPT baseline. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"},{"label":"Inputs","value":"immune tissue","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["immune tissue","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["single-cell-residual-geometry-2026"],"source_locator":"Table 4, Immune row, scGPT > +geom AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for immune tissue have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-9f62e739c6371e","kind":"benchmark","name":"human core-promoter classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["human core-promoter classification"]},"source_ids":["eden-genomic-classification-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-8e9488896becd4"}],"attributes":{"entity_level":"task","version":null,"task":"human core-promoter classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests human core-promoter classification using GUE H-CPD.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: DNABERT-2 comparator in consolidated H-CPD table; rerun provenance not explicit. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"},{"label":"Inputs","value":"GUE H-CPD","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"},{"label":"Assessment","value":"MCC","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"},{"label":"Recorded split or evaluation setting","value":"paper evaluation","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"},{"text":"Some comparator provenance is quoted or unresolved. Do not treat copied comparisons as independent evaluations or assume identical protocols.","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"}],"diagram":{"title":"Reported evaluation outline","steps":["GUE H-CPD","Recorded fitting or scoring procedure","Assess MCC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["eden-genomic-classification-2026"],"source_locator":"Table 5, DNABERT-2 row, H-CPD (MCC) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for GUE H-CPD have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-9f9ab0090f6522","kind":"benchmark","name":"Natural vs artificial microbial genome sequence","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Natural vs artificial microbial genome sequence"]},"source_ids":["genomeocean-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-0c3ac7efe99c37"}],"attributes":{"entity_level":"task","version":null,"task":"Natural vs artificial microbial genome sequence","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Natural vs artificial microbial genome sequence using GenomeOcean natural/artificial sequence test.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Source reports natural-versus-artificial sequence classification. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column; Table 2, DNABERT-2 row, F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column; Table 2, DNABERT-2 row, F1 column"},{"label":"Inputs","value":"GenomeOcean natural/artificial sequence test","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column; Table 2, DNABERT-2 row, F1 column"},{"label":"Assessment","value":"F1","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column; Table 2, DNABERT-2 row, F1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column; Table 2, DNABERT-2 row, F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column; Table 2, DNABERT-2 row, F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column; Table 2, DNABERT-2 row, F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["GenomeOcean natural/artificial sequence test","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["genomeocean-2025"],"source_locator":"Table 2, GenomeOcean row, F1 column; Table 2, DNABERT-2 row, F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for GenomeOcean natural/artificial sequence test have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-a1151e386a3d3f","kind":"benchmark","name":"Multi-species prokaryotic promoter detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Multi-species prokaryotic promoter detection"]},"source_ids":["ipromp-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-e0f34dcaa1ba3b"}],"attributes":{"entity_level":"task","version":null,"task":"Multi-species prokaryotic promoter detection","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Multi-species prokaryotic promoter detection using 23 independent prokaryotic promoter test sets.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Average over independent testing sets.; Average over the same independent testing sets. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column; Table 2, Prompt row, AUC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column; Table 2, Prompt row, AUC column"},{"label":"Inputs","value":"23 independent prokaryotic promoter test sets","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column; Table 2, Prompt row, AUC column"},{"label":"Assessment","value":"Mean AUC","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column; Table 2, Prompt row, AUC column"},{"label":"Recorded split or evaluation setting","value":"independent test","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column; Table 2, Prompt row, AUC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column; Table 2, Prompt row, AUC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column; Table 2, Prompt row, AUC column"}],"diagram":{"title":"Reported evaluation outline","steps":["23 independent prokaryotic promoter test sets","Recorded fitting or scoring procedure","Assess Mean AUC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["ipromp-2025"],"source_locator":"Table 2, iPro-MP row, AUC column; Table 2, Prompt row, AUC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for 23 independent prokaryotic promoter test sets have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-a2bf7ddbc71d23","kind":"benchmark","name":"RNA secondary-structure prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary-structure prediction"]},"source_ids":["ernie-rna-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-abdfba8cce7486"}],"attributes":{"entity_level":"task","version":null,"task":"RNA secondary-structure prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests RNA secondary-structure prediction using bpRNA-new.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: zero-shot attention-derived base-pair prediction. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"},{"label":"Inputs","value":"bpRNA-new","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"},{"label":"Assessment","value":"binary F1","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"},{"label":"Recorded split or evaluation setting","value":"cross-family test","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"}],"diagram":{"title":"Reported evaluation outline","steps":["bpRNA-new","Recorded fitting or scoring procedure","Assess binary F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["ernie-rna-2025"],"source_locator":"Table 2, ERNIE-RNA zero-shot row, bpRNA-new F1-Score (binary) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for bpRNA-new have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-a2c37b8c420bc3","kind":"benchmark","name":"CAMI II superkingdom read classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["CAMI II superkingdom read classification"]},"source_ids":["ncd-metagenomics-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-beb4f5da29da0a"}],"attributes":{"entity_level":"task","version":null,"task":"CAMI II superkingdom read classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests CAMI II superkingdom read classification using CAMI II Sample_0 10,000-read subsample.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Superkingdom-level macro-averaged F1; NCD assigns every read. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column"},{"label":"Inputs","value":"CAMI II Sample_0 10,000-read subsample","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column"},{"label":"Assessment","value":"Macro F1","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["CAMI II Sample_0 10,000-read subsample","Recorded fitting or scoring procedure","Assess Macro F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["ncd-metagenomics-2026"],"source_locator":"Table 5, NCD Superkingdom row, F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for CAMI II Sample_0 10,000-read subsample have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-a5141363b0ee45","kind":"benchmark","name":"Zero-shot variant effect prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["Zero-shot variant effect prediction"]},"source_ids":["pst-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-bd9255afb783d6"}],"attributes":{"entity_level":"task","version":null,"task":"Zero-shot variant effect prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Zero-shot variant effect prediction using ProteinShake VEP datasets.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Zero-shot VEP; paper averages absolute Spearman correlations. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column; Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column; Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"},{"label":"Inputs","value":"ProteinShake VEP datasets","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column; Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"},{"label":"Assessment","value":"Mean |Spearman rho|","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column; Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column; Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column; Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column; Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"}],"diagram":{"title":"Reported evaluation outline","steps":["ProteinShake VEP datasets","Recorded fitting or scoring procedure","Assess Mean |Spearman rho|"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["pst-2025"],"source_locator":"Table 2, PST row, Zero-shot VEP Mean |ρ| column; Table 2, ESM-2 row, Zero-shot VEP Mean |ρ| column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for ProteinShake VEP datasets have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-a7803ecf7708cc","kind":"benchmark","name":"Protein–ligand virtual screening","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand virtual screening"]},"source_ids":["nmdn-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-065b9fcc8da573"}],"attributes":{"entity_level":"task","version":null,"task":"Protein–ligand virtual screening","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Protein–ligand virtual screening using CASF-2016 blind docked poses.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: NMDN scoring on DiffDock-NMDN blind docked poses; not ligand-pose RMSD.; Vina scoring on the same DiffDock-NMDN blind docked poses; not ligand-pose RMSD. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column; Table 2, Vina scoring row, success rate (%) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column; Table 2, Vina scoring row, success rate (%) column"},{"label":"Inputs","value":"CASF-2016 blind docked poses","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column; Table 2, Vina scoring row, success rate (%) column"},{"label":"Assessment","value":"Forward-screening success rate","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column; Table 2, Vina scoring row, success rate (%) column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column; Table 2, Vina scoring row, success rate (%) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column; Table 2, Vina scoring row, success rate (%) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column; Table 2, Vina scoring row, success rate (%) column"}],"diagram":{"title":"Reported evaluation outline","steps":["CASF-2016 blind docked poses","Recorded fitting or scoring procedure","Assess Forward-screening success rate"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["nmdn-2025"],"source_locator":"Table 2, DiffDock-NMDN / NMDN row, success rate (%) column; Table 2, Vina scoring row, success rate (%) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for CASF-2016 blind docked poses have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-a78312d5df6dad","kind":"benchmark","name":"Protein–ligand binding affinity scoring","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity scoring"]},"source_ids":["akscore-2020"],"links":[{"relation":"dataset","target_id":"reported-dataset-f18fcc23dfa798"}],"attributes":{"entity_level":"task","version":null,"task":"Protein–ligand binding affinity scoring","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Protein–ligand binding affinity scoring using CASF-2016.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: CASF-2016 scoring-power evaluation. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column; Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column; Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"},{"label":"Inputs","value":"CASF-2016","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column; Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"},{"label":"Assessment","value":"Pearson R","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column; Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column; Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column; Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column; Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"}],"diagram":{"title":"Reported evaluation outline","steps":["CASF-2016","Recorded fitting or scoring procedure","Assess Pearson R"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["akscore-2020"],"source_locator":"Table 2, AK-score-ensemble / learning rate 0.0007 row, Scoring Pearson (R) column; Table 2, AK-score-single / learning rate 0.0007 row, Scoring Pearson (R) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for CASF-2016 have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-ac191e878dff5e","kind":"benchmark","name":"transcription-factor DNA binding-site prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["transcription-factor DNA binding-site prediction"]},"source_ids":["transbind-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-034c60a2dabc73"}],"attributes":{"entity_level":"task","version":null,"task":"transcription-factor DNA binding-site prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests transcription-factor DNA binding-site prediction using genome-wide TF binding sites.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Integrates protein and DNA embeddings for TFBS prediction on paper test dataset. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"},{"label":"Inputs","value":"genome-wide TF binding sites","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"},{"label":"Recorded split or evaluation setting","value":"test","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["genome-wide TF binding sites","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["transbind-2026"],"source_locator":"Table 2, TransBind row, AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for genome-wide TF binding sites have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-b00a636d1ed8d9","kind":"benchmark","name":"RNA-small-molecule binding-site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA-small-molecule binding-site prediction"]},"source_ids":["rlsite-rna-binding-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-1c7f8ebb1968d9"}],"attributes":{"entity_level":"task","version":null,"task":"RNA-small-molecule binding-site prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests RNA-small-molecule binding-site prediction using T18.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: RNA language-model plus graph-attention classifier. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"},{"label":"Inputs","value":"T18","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"},{"label":"Assessment","value":"AUC","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"},{"label":"Recorded split or evaluation setting","value":"paper evaluation","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"}],"diagram":{"title":"Reported evaluation outline","steps":["T18","Recorded fitting or scoring procedure","Assess AUC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["rlsite-rna-binding-2025"],"source_locator":"Table 1, RLsite row, T18 AUC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for T18 have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-b181ed450cdd41","kind":"benchmark","name":"protein-small molecule binding-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-small molecule binding-site prediction"]},"source_ids":["clape-smb-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-701d910b02d25c"}],"attributes":{"entity_level":"task","version":null,"task":"protein-small molecule binding-site prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests protein-small molecule binding-site prediction using SJC.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Contrastive CLAPE-SMB binding-site predictor with ESM-2 feature extractor. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"},{"label":"Inputs","value":"SJC","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["SJC","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["clape-smb-2024"],"source_locator":"Table 5, ESM-2 / SJC row, AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for SJC have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-b46b7b839bff93","kind":"benchmark","name":"PBMC cell-type classification","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["PBMC cell-type classification"]},"source_ids":["scalr-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-33a41fe5fc66cf"}],"attributes":{"entity_level":"task","version":null,"task":"PBMC cell-type classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests PBMC cell-type classification using PBMCs-BS.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: All features and samples from PBMCs-BS.; All features and samples from PBMCs-BS; comparison pipeline combines scVI and scANVI. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column; Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column; Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"},{"label":"Inputs","value":"PBMCs-BS","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column; Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"},{"label":"Assessment","value":"Cell-type accuracy","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column; Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column; Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column; Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column; Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"}],"diagram":{"title":"Reported evaluation outline","steps":["PBMCs-BS","Recorded fitting or scoring procedure","Assess Cell-type accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["scalr-2025"],"source_locator":"Table 2, scaLR row, Cell type Accuracy column; Table 2, Svi-tools (scVI & scANVI) row, Cell type Accuracy column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for PBMCs-BS have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-b9199a30a0bcb2","kind":"benchmark","name":"regulatory-variant scoring","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory-variant scoring"]},"source_ids":["arsenal-regulatory-dna-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-158b121281b650"}],"attributes":{"entity_level":"task","version":null,"task":"regulatory-variant scoring","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests regulatory-variant scoring using Yoruban LCL dsQTLs.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Supervised ChromBPNet variant scoring with ARSENAL motif-discovery regularization. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"},{"label":"Inputs","value":"Yoruban LCL dsQTLs","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["Yoruban LCL dsQTLs","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["arsenal-regulatory-dna-2026"],"source_locator":"Table 1, Yoruban LCL dsQTLs section, ARSENAL+ChromBPNet row, AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Yoruban LCL dsQTLs have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-bf513ed6db92c5","kind":"benchmark","name":"Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz-stereochemistry-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-5afaefb87c8a94"}],"attributes":{"entity_level":"task","version":null,"task":"Protein–ligand pose prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Protein–ligand pose prediction using PLINDER-L95.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: All entries; authors note this dataset contains structures seen during model training.; All entries; rigid-protein docking comparator; authors note this dataset contains structures seen during model training. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column; Table 1, DiffDock row, Ligand RMSD (Å) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column; Table 1, DiffDock row, Ligand RMSD (Å) column"},{"label":"Inputs","value":"PLINDER-L95","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column; Table 1, DiffDock row, Ligand RMSD (Å) column"},{"label":"Assessment","value":"Median ligand RMSD","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column; Table 1, DiffDock row, Ligand RMSD (Å) column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column; Table 1, DiffDock row, Ligand RMSD (Å) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column; Table 1, DiffDock row, Ligand RMSD (Å) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column; Table 1, DiffDock row, Ligand RMSD (Å) column"}],"diagram":{"title":"Reported evaluation outline","steps":["PLINDER-L95","Recorded fitting or scoring procedure","Assess Median ligand RMSD"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["boltz-stereochemistry-2025"],"source_locator":"Table 1, Boltz-1 row, Ligand RMSD (Å) column; Table 1, DiffDock row, Ligand RMSD (Å) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for PLINDER-L95 have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-c04bb5ee6ecea6","kind":"benchmark","name":"Protein–ligand pose prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand pose prediction"]},"source_ids":["boltz1-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-e45a5a140888ee"}],"attributes":{"entity_level":"task","version":null,"task":"Protein–ligand pose prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Protein–ligand pose prediction using Boltz-1 structure test set.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Highest-confidence pose from five samples; precomputed MSAs up to 4,096 sequences. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"},{"label":"Inputs","value":"Boltz-1 structure test set","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"},{"label":"Assessment","value":"Top-1 ligand RMSD <2 Å rate","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["Boltz-1 structure test set","Recorded fitting or scoring procedure","Assess Top-1 ligand RMSD <2 Å rate"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["boltz1-2025"],"source_locator":"Table 1, 3 recycling rounds / 200 steps row, L-RMSD <2Å top-1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Boltz-1 structure test set have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-c40dac20d9af66","kind":"benchmark","name":"extremely long RNA species classification","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["extremely long RNA species classification"]},"source_ids":["birna-bert-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-ebc3f5fda43972"}],"attributes":{"entity_level":"task","version":null,"task":"extremely long RNA species classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests extremely long RNA species classification using extremely long-sequence species classification.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: adaptive tokenization on full-length long RNA sequences. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"},{"label":"Inputs","value":"extremely long-sequence species classification","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"},{"label":"Assessment","value":"F1","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"},{"label":"Recorded split or evaluation setting","value":"paper evaluation","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"}],"diagram":{"title":"Reported evaluation outline","steps":["extremely long-sequence species classification","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["birna-bert-2025"],"source_locator":"Table 2, BiRNA-BERT row, F1 Score column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for extremely long-sequence species classification have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-c4a578065f44b2","kind":"benchmark","name":"protein function annotation","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein function annotation"]},"source_ids":["spin-protein-function-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-dba1707164d296"}],"attributes":{"entity_level":"task","version":null,"task":"protein function annotation","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests protein function annotation using TRX.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: frozen ESM2-35M backbone in SPIN. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"},{"label":"Inputs","value":"TRX","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"},{"label":"Assessment","value":"F1 macro-weighted","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"},{"label":"Recorded split or evaluation setting","value":"test set","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"}],"diagram":{"title":"Reported evaluation outline","steps":["TRX","Recorded fitting or scoring procedure","Assess F1 macro-weighted"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["spin-protein-function-2026"],"source_locator":"Table 1, ESM2-35M Test row, F1_m-w column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for TRX have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-c7a8a372f77886","kind":"benchmark","name":"protein variant fitness prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein variant fitness prediction"]},"source_ids":["esm2-ofs-fitness-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-9c186c8f4ed3f4"}],"attributes":{"entity_level":"task","version":null,"task":"protein variant fitness prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests protein variant fitness prediction using ProteinGym substitutions.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: authors’ zero-shot ESM2 OFS pseudo-perplexity evaluation; aggregate mean across ProteinGym substitution assays. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"},{"label":"Inputs","value":"ProteinGym substitutions","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"},{"label":"Assessment","value":"Spearman rho","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"},{"label":"Recorded split or evaluation setting","value":"aggregate across assays","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"}],"diagram":{"title":"Reported evaluation outline","steps":["ProteinGym substitutions","Recorded fitting or scoring procedure","Assess Spearman rho"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["esm2-ofs-fitness-2025"],"source_locator":"Published PDF page 6 (033014-6), Table I, ESM2: OFS PP row, Aggregate mean column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for ProteinGym substitutions have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-c7ce06b753b8b6","kind":"benchmark","name":"non-coding RNA pairwise interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["non-coding RNA pairwise interaction prediction"]},"source_ids":["cupid-rna-interactions-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-32ccef507a1dd7"}],"attributes":{"entity_level":"task","version":null,"task":"non-coding RNA pairwise interaction prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests non-coding RNA pairwise interaction prediction using ncRNA interaction pairs.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Data augmentation with average pooling for molecule-level ncRNA embeddings. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"},{"label":"Inputs","value":"ncRNA interaction pairs","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["ncRNA interaction pairs","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["cupid-rna-interactions-2026"],"source_locator":"Table 1, CUPID > Data-aug-Avg row, AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for ncRNA interaction pairs have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-c98e91ffc7247d","kind":"benchmark","name":"CATH superfamily annotation","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["CATH superfamily annotation"]},"source_ids":["cathe2-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-6e0c28dfde7337"}],"attributes":{"entity_level":"task","version":null,"task":"CATH superfamily annotation","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests CATH superfamily annotation using CATH superfamily benchmark.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: amino-acid and structural alphabet embedding classifier. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"},{"label":"Inputs","value":"CATH superfamily benchmark","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"},{"label":"Assessment","value":"F1","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"},{"label":"Recorded split or evaluation setting","value":"paper evaluation","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"}],"diagram":{"title":"Reported evaluation outline","steps":["CATH superfamily benchmark","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["cathe2-2025"],"source_locator":"Table 3, ProstT5 full row, F1 score column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for CATH superfamily benchmark have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-c9d2a6435979e9","kind":"benchmark","name":"G-quadruplex classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["G-quadruplex classification"]},"source_ids":["quadruplex-llm-benchmark-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-9e9d18bc5bfb8b"}],"attributes":{"entity_level":"task","version":null,"task":"G-quadruplex classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests G-quadruplex classification using KEx.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Pretrained model evaluated on KEx as reported in Table 5. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column; Table 5, Caduceus (8 M) row, Accuracy column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column; Table 5, Caduceus (8 M) row, Accuracy column"},{"label":"Inputs","value":"KEx","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column; Table 5, Caduceus (8 M) row, Accuracy column"},{"label":"Assessment","value":"Accuracy","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column; Table 5, Caduceus (8 M) row, Accuracy column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column; Table 5, Caduceus (8 M) row, Accuracy column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column; Table 5, Caduceus (8 M) row, Accuracy column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column; Table 5, Caduceus (8 M) row, Accuracy column"}],"diagram":{"title":"Reported evaluation outline","steps":["KEx","Recorded fitting or scoring procedure","Assess Accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["quadruplex-llm-benchmark-2025"],"source_locator":"Table 5, DNABERT-2 (117 M) row, Accuracy column; Table 5, Caduceus (8 M) row, Accuracy column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for KEx have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-cd127e56fb1f04","kind":"benchmark","name":"regulatory sequence classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory sequence classification"]},"source_ids":["genomic-tokenizer-selection-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-0bba1c9a7ae410"}],"attributes":{"entity_level":"task","version":null,"task":"regulatory sequence classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests regulatory sequence classification using genomic benchmark categories.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: task-category MCC across benchmark datasets. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"},{"label":"Inputs","value":"genomic benchmark categories","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"},{"label":"Assessment","value":"MCC","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"},{"label":"Recorded split or evaluation setting","value":"paper benchmark summary","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"}],"diagram":{"title":"Reported evaluation outline","steps":["genomic benchmark categories","Recorded fitting or scoring procedure","Assess MCC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["genomic-tokenizer-selection-2025"],"source_locator":"Table 2, Regulatory row, Caduceus (char) MCC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for genomic benchmark categories have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-cdbee1c9285568","kind":"benchmark","name":"regulatory element identification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["regulatory element identification"]},"source_ids":["dart-eval-regulatory-2024"],"links":[{"relation":"dataset","target_id":"reported-dataset-b6ce37ba678d39"}],"attributes":{"entity_level":"task","version":null,"task":"regulatory element identification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests regulatory element identification using DART-Eval cCREs versus matched shuffled controls.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: zero-shot likelihood ranking: higher likelihood for cCRE than matched control. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"},{"label":"Inputs","value":"DART-Eval cCREs versus matched shuffled controls","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"},{"label":"Assessment","value":"accuracy","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"},{"label":"Recorded split or evaluation setting","value":"paper evaluation","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"}],"diagram":{"title":"Reported evaluation outline","steps":["DART-Eval cCREs versus matched shuffled controls","Recorded fitting or scoring procedure","Assess accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["dart-eval-regulatory-2024"],"source_locator":"Table 3 (PDF page 5), DNABERT-2 row, Zero-Shot Accuracy column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for DART-Eval cCREs versus matched shuffled controls have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-d1c46526c39983","kind":"benchmark","name":"Physically valid protein–ligand pose selection","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Physically valid protein–ligand pose selection"]},"source_ids":["molas-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-4b6c13924d4256"}],"attributes":{"entity_level":"task","version":null,"task":"Physically valid protein–ligand pose selection","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Physically valid protein–ligand pose selection using PoseBusters.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Averaged five-fold algorithm-selection performance on PoseBusters; joint RMSD and validity criterion.; Single best solver baseline under the same averaged five-fold selection test. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column; Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column; Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"},{"label":"Inputs","value":"PoseBusters","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column; Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"},{"label":"Assessment","value":"RMSD ≤1 Å and PB-valid success","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column; Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column; Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column; Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column; Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"}],"diagram":{"title":"Reported evaluation outline","steps":["PoseBusters","Recorded fitting or scoring procedure","Assess RMSD ≤1 Å and PB-valid success"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["molas-2026"],"source_locator":"Table 3, PoseBusters / Mixed / AutoDock row, MolAS success column; Table 3, PoseBusters / Mixed / AutoDock row, SBS success column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for PoseBusters have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-d3fd502fdc2b38","kind":"benchmark","name":"pathogen detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["pathogen detection"]},"source_ids":["metagenomic-pathogens-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-1c4c71078ffe01"}],"attributes":{"entity_level":"task","version":null,"task":"pathogen detection","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests pathogen detection using MetaHIT.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Taxonomy-constrained inference network with hierarchical taxonomy representation. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"},{"label":"Inputs","value":"MetaHIT","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"},{"label":"Assessment","value":"F1","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"}],"diagram":{"title":"Reported evaluation outline","steps":["MetaHIT","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["metagenomic-pathogens-2025"],"source_locator":"Table 3, MetaHIT dataset section, TCINet + HTRS (Ours) row, F1-score column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for MetaHIT have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-d5f897ab0f6f67","kind":"benchmark","name":"Ligand potency prediction using generated poses","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Ligand potency prediction using generated poses"]},"source_ids":["mpro-pose-affinity-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-235520c84b737f"}],"attributes":{"entity_level":"task","version":null,"task":"Ligand potency prediction using generated poses","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Ligand potency prediction using generated poses using SARS-CoV-2 Mpro ligands.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Potency prediction using Boltz-2 ligand-pose generation protocol; see paper scoring pipeline.; Potency prediction using DiffDock ligand-pose generation plus paper scoring pipeline; not a native DiffDock affinity score. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column; Table 3, DiffDock row, Pearson’s R column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column; Table 3, DiffDock row, Pearson’s R column"},{"label":"Inputs","value":"SARS-CoV-2 Mpro ligands","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column; Table 3, DiffDock row, Pearson’s R column"},{"label":"Assessment","value":"Pearson R","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column; Table 3, DiffDock row, Pearson’s R column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column; Table 3, DiffDock row, Pearson’s R column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column; Table 3, DiffDock row, Pearson’s R column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column; Table 3, DiffDock row, Pearson’s R column"}],"diagram":{"title":"Reported evaluation outline","steps":["SARS-CoV-2 Mpro ligands","Recorded fitting or scoring procedure","Assess Pearson R"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["mpro-pose-affinity-2025"],"source_locator":"Table 3, Boltz-2 row, Pearson’s R column; Table 3, DiffDock row, Pearson’s R column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for SARS-CoV-2 Mpro ligands have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-d6018ca598e525","kind":"benchmark","name":"Donor-aware age-class prediction","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Donor-aware age-class prediction"]},"source_ids":["single-cell-aging-probes-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-571ce000cd74b9"}],"attributes":{"entity_level":"task","version":null,"task":"Donor-aware age-class prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Donor-aware age-class prediction using AIDA v2 PBMC cohort.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Same donor-aware splits and logistic-regression probe as expression PCA; text names Geneformer as best model on AIDA v2.; Fifty-component gene-expression PCA with the same donor-aware probe splits. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column; Table 2, AIDA v2 row, Gene-expr BA column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column; Table 2, AIDA v2 row, Gene-expr BA column"},{"label":"Inputs","value":"AIDA v2 PBMC cohort","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column; Table 2, AIDA v2 row, Gene-expr BA column"},{"label":"Assessment","value":"Balanced accuracy","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column; Table 2, AIDA v2 row, Gene-expr BA column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column; Table 2, AIDA v2 row, Gene-expr BA column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column; Table 2, AIDA v2 row, Gene-expr BA column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column; Table 2, AIDA v2 row, Gene-expr BA column"}],"diagram":{"title":"Reported evaluation outline","steps":["AIDA v2 PBMC cohort","Recorded fitting or scoring procedure","Assess Balanced accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["single-cell-aging-probes-2026"],"source_locator":"Table 2, AIDA v2 row, scFM BA ± SD column; Table 2, AIDA v2 row, Gene-expr BA column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for AIDA v2 PBMC cohort have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-d635fc6c281a27","kind":"benchmark","name":"A-to-I RNA editing site prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["A-to-I RNA editing site prediction"]},"source_ids":["adar-gpt-editing-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-236eaa4e55147f"}],"attributes":{"entity_level":"task","version":null,"task":"A-to-I RNA editing site prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests A-to-I RNA editing site prediction using liver editing sites.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Curriculum plus 15% fine-tuning; 201-nt sequence windows; decision threshold 0.5. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"},{"label":"Inputs","value":"liver editing sites","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"},{"label":"Assessment","value":"F1","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"},{"label":"Recorded split or evaluation setting","value":"15% validation set","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["liver editing sites","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["adar-gpt-editing-2026"],"source_locator":"Table 2, Adar-GPT (continual) row, F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for liver editing sites have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-d7e6274011946e","kind":"benchmark","name":"mRNA-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["mRNA-protein interaction prediction"]},"source_ids":["mrna-protein-diversity-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-2e87449871ca47"}],"attributes":{"entity_level":"task","version":null,"task":"mRNA-protein interaction prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests mRNA-protein interaction prediction using mRNA-RBP pairs.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: LLM encoding of protein partner; RBP-aware partition tests generalization to unseen protein diversity. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"},{"label":"Inputs","value":"mRNA-RBP pairs","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"},{"label":"Assessment","value":"AUROC","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"},{"label":"Recorded split or evaluation setting","value":"RBP-aware test set","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"}],"diagram":{"title":"Reported evaluation outline","steps":["mRNA-RBP pairs","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["mrna-protein-diversity-2026"],"source_locator":"Table 2, RBP-aware test set row, auROC (%) column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for mRNA-RBP pairs have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-d81be76396e644","kind":"benchmark","name":"Protein–ligand binding affinity prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Protein–ligand binding affinity prediction"]},"source_ids":["deelig-2021"],"links":[{"relation":"dataset","target_id":"reported-dataset-327cfcdae0c937"}],"attributes":{"entity_level":"task","version":null,"task":"Protein–ligand binding affinity prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Protein–ligand binding affinity prediction using PDBbind core v2016.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Source paper reports DEELIG on PDBbind core set.; Source table compiles a previously published comparator; protocol equivalence is not established. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column; Table 2, TOPBP (Complex) row, PDBbind v2016 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column; Table 2, TOPBP (Complex) row, PDBbind v2016 column"},{"label":"Inputs","value":"PDBbind core v2016","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column; Table 2, TOPBP (Complex) row, PDBbind v2016 column"},{"label":"Assessment","value":"Pearson R","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column; Table 2, TOPBP (Complex) row, PDBbind v2016 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column; Table 2, TOPBP (Complex) row, PDBbind v2016 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column; Table 2, TOPBP (Complex) row, PDBbind v2016 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column; Table 2, TOPBP (Complex) row, PDBbind v2016 column"},{"text":"Some comparator provenance is quoted or unresolved. Do not treat copied comparisons as independent evaluations or assume identical protocols.","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column; Table 2, TOPBP (Complex) row, PDBbind v2016 column"}],"diagram":{"title":"Reported evaluation outline","steps":["PDBbind core v2016","Recorded fitting or scoring procedure","Assess Pearson R"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["deelig-2021"],"source_locator":"Table 2, DEELIG row, PDBbind v2016 column; Table 2, TOPBP (Complex) row, PDBbind v2016 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for PDBbind core v2016 have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-d82b6284f3f431","kind":"benchmark","name":"Cross-platform scATAC cell-type annotation","description":"","status":"needs_review","facets":{"areas":["cells-tissues"],"tasks":["Cross-platform scATAC cell-type annotation"]},"source_ids":["scatac-llmda-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-7fc59ce4c0ceaa"}],"attributes":{"entity_level":"task","version":null,"task":"Cross-platform scATAC cell-type annotation","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Cross-platform scATAC cell-type annotation using MosA1 reference → WholeBrainA query.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Cross-platform reference-query cell-type annotation.; Cross-platform reference-query comparator. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column; Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column; Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"},{"label":"Inputs","value":"MosA1 reference → WholeBrainA query","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column; Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"},{"label":"Assessment","value":"F1","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column; Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column; Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column; Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column; Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["MosA1 reference → WholeBrainA query","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["scatac-llmda-2026"],"source_locator":"Table 2, scLLMDA row, Ref: MosA1 / Q: WholeBrainA F1 column; Table 2, MINGLE row, Ref: MosA1 / Q: WholeBrainA F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for MosA1 reference → WholeBrainA query have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-dc82fcbfb44935","kind":"benchmark","name":"RNA secondary structure","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA secondary structure"]},"source_ids":["bpfold-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-8317793f18b026"}],"attributes":{"entity_level":"task","version":null,"task":"RNA secondary structure","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests RNA secondary structure using PDB RNA set.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Family-wise evaluation of canonical base-pair predictions. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column; Table 2, RNAfold row, PDB F1 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column; Table 2, RNAfold row, PDB F1 column"},{"label":"Inputs","value":"PDB RNA set","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column; Table 2, RNAfold row, PDB F1 column"},{"label":"Assessment","value":"F1","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column; Table 2, RNAfold row, PDB F1 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column; Table 2, RNAfold row, PDB F1 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column; Table 2, RNAfold row, PDB F1 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column; Table 2, RNAfold row, PDB F1 column"}],"diagram":{"title":"Reported evaluation outline","steps":["PDB RNA set","Recorded fitting or scoring procedure","Assess F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["bpfold-2025"],"source_locator":"Table 2, BPfold row, PDB F1 column; Table 2, RNAfold row, PDB F1 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for PDB RNA set have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-dd001540e0f4ec","kind":"benchmark","name":"Genome-wide prophage detection","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Genome-wide prophage detection"]},"source_ids":["lambda-prophage-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-1b4f6ea24c0587"}],"attributes":{"entity_level":"task","version":null,"task":"Genome-wide prophage detection","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Genome-wide prophage detection using LAMBDA genome-wide prophage test.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Genomic language model fine-tuned for prophage detection; genome-wide evaluation.; Traditional specialist comparator; genome-wide evaluation. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column; Table 5, geNomad row, MCC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column; Table 5, geNomad row, MCC column"},{"label":"Inputs","value":"LAMBDA genome-wide prophage test","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column; Table 5, geNomad row, MCC column"},{"label":"Assessment","value":"MCC","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column; Table 5, geNomad row, MCC column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column; Table 5, geNomad row, MCC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column; Table 5, geNomad row, MCC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column; Table 5, geNomad row, MCC column"}],"diagram":{"title":"Reported evaluation outline","steps":["LAMBDA genome-wide prophage test","Recorded fitting or scoring procedure","Assess MCC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["lambda-prophage-2026"],"source_locator":"Table 5, EVO2 row, MCC column; Table 5, geNomad row, MCC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for LAMBDA genome-wide prophage test have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-dec9e0f5e3da2a","kind":"benchmark","name":"Intrinsically disordered protein ensemble docking","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Intrinsically disordered protein ensemble docking"]},"source_ids":["ensemble-idp-docking-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-00201f65c32f6d"}],"attributes":{"entity_level":"task","version":null,"task":"Intrinsically disordered protein ensemble docking","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Intrinsically disordered protein ensemble docking using α-synuclein Ligand 47 MD ensemble.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Fraction of docked frames within 3 Å of MD-observed bound pose; holo protein ensemble. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column; Table 2, Ligand 47 row, DiffDock Holo Docking column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column; Table 2, Ligand 47 row, DiffDock Holo Docking column"},{"label":"Inputs","value":"α-synuclein Ligand 47 MD ensemble","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column; Table 2, Ligand 47 row, DiffDock Holo Docking column"},{"label":"Assessment","value":"Docked frames best-matched RMSD <3 Å","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column; Table 2, Ligand 47 row, DiffDock Holo Docking column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column; Table 2, Ligand 47 row, DiffDock Holo Docking column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column; Table 2, Ligand 47 row, DiffDock Holo Docking column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column; Table 2, Ligand 47 row, DiffDock Holo Docking column"}],"diagram":{"title":"Reported evaluation outline","steps":["α-synuclein Ligand 47 MD ensemble","Recorded fitting or scoring procedure","Assess Docked frames best-matched RMSD <3 Å"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["ensemble-idp-docking-2025"],"source_locator":"Table 2, Ligand 47 row, AutoDock Vina Holo Docking column; Table 2, Ligand 47 row, DiffDock Holo Docking column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for α-synuclein Ligand 47 MD ensemble have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-df18c710f45213","kind":"benchmark","name":"RNA sequence design","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["RNA sequence design"]},"source_ids":["r3design-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-71614d99b3099f"}],"attributes":{"entity_level":"task","version":null,"task":"RNA sequence design","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests RNA sequence design using Rfam.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Tertiary-structure-conditioned RNA sequence design; external Rfam assessment. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"},{"label":"Inputs","value":"Rfam","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"},{"label":"Assessment","value":"sequence recovery","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"},{"label":"Recorded split or evaluation setting","value":"external","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"}],"diagram":{"title":"Reported evaluation outline","steps":["Rfam","Recorded fitting or scoring procedure","Assess sequence recovery"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["r3design-2025"],"source_locator":"Table 3, R3Design row, Recovery (%) > Rfam column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Rfam have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-dfa8f2285dbfa5","kind":"benchmark","name":"protein-protein interaction prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein interaction prediction"]},"source_ids":["gsmformer-ppi-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-07d355c146be1f"}],"attributes":{"entity_level":"task","version":null,"task":"protein-protein interaction prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests protein-protein interaction prediction using paper PPI test set.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: ProstT5 embeddings as graph node features. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"},{"label":"Inputs","value":"paper PPI test set","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"},{"label":"Recorded split or evaluation setting","value":"test set","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["paper PPI test set","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["gsmformer-ppi-2026"],"source_locator":"Table 6, ProstT5 embedding row, AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for paper PPI test set have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-e2009c35eabd69","kind":"benchmark","name":"microbiome disease-state classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["microbiome disease-state classification"]},"source_ids":["mdl4microbiome-2022"],"links":[{"relation":"dataset","target_id":"reported-dataset-bd3f98e2eeb5d3"}],"attributes":{"entity_level":"task","version":null,"task":"microbiome disease-state classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests microbiome disease-state classification using CRC microbiome cohort.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Multimodal deep learning model on colorectal-cancer versus healthy microbiome samples. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"},{"label":"Inputs","value":"CRC microbiome cohort","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"},{"label":"Assessment","value":"accuracy","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"}],"diagram":{"title":"Reported evaluation outline","steps":["CRC microbiome cohort","Recorded fitting or scoring procedure","Assess accuracy"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["mdl4microbiome-2022"],"source_locator":"Table 3, CRC row, MDL4Microbiome column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for CRC microbiome cohort have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-e5c34f686ac403","kind":"benchmark","name":"E. coli sigma70 promoter prediction","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["E. coli sigma70 promoter prediction"]},"source_ids":["cyaprombert-2022"],"links":[{"relation":"dataset","target_id":"reported-dataset-a1da4a37eb46a5"}],"attributes":{"entity_level":"task","version":null,"task":"E. coli sigma70 promoter prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests E. coli sigma70 promoter prediction using Independent E. coli sigma70 test dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: BERT-base with 1bp tokenizer; 110 promoters and 108 non-promoters.; Compared on the same independent test dataset; 110 promoters and 108 non-promoters. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column; TABLE 3, iPro70-FMWin row, F1 score Promoter column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column; TABLE 3, iPro70-FMWin row, F1 score Promoter column"},{"label":"Inputs","value":"Independent E. coli sigma70 test dataset","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column; TABLE 3, iPro70-FMWin row, F1 score Promoter column"},{"label":"Assessment","value":"Promoter-class F1","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column; TABLE 3, iPro70-FMWin row, F1 score Promoter column"},{"label":"Recorded split or evaluation setting","value":"independent test","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column; TABLE 3, iPro70-FMWin row, F1 score Promoter column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column; TABLE 3, iPro70-FMWin row, F1 score Promoter column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column; TABLE 3, iPro70-FMWin row, F1 score Promoter column"}],"diagram":{"title":"Reported evaluation outline","steps":["Independent E. coli sigma70 test dataset","Recorded fitting or scoring procedure","Assess Promoter-class F1"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["cyaprombert-2022"],"source_locator":"TABLE 3, Eco70PromBERT (BERT-base + 1bp tokenizer) row, F1 score Promoter column; TABLE 3, iPro70-FMWin row, F1 score Promoter column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Independent E. coli sigma70 test dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-ed3dd3b83c4505","kind":"benchmark","name":"ClinVar 3-prime UTR variant classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["ClinVar 3-prime UTR variant classification"]},"source_ids":["phylogpn-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-a28180d33f7a23"}],"attributes":{"entity_level":"task","version":null,"task":"ClinVar 3-prime UTR variant classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests ClinVar 3-prime UTR variant classification using ClinVar 3-prime UTR variants.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: log-likelihood-ratio scoring. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"},{"label":"Inputs","value":"ClinVar 3-prime UTR variants","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"},{"label":"Recorded split or evaluation setting","value":"paper evaluation","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["ClinVar 3-prime UTR variants","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["phylogpn-2025"],"source_locator":"Table 1, 3-prime UTR row, PhyloGPN AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for ClinVar 3-prime UTR variants have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-ee34721cf55590","kind":"benchmark","name":"gene fusion breakpoint classification","description":"","status":"needs_review","facets":{"areas":["dna-genomes"],"tasks":["gene fusion breakpoint classification"]},"source_ids":["fusion-breakpoint-foundation-models-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-f6922a9744ba27"}],"attributes":{"entity_level":"task","version":null,"task":"gene fusion breakpoint classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests gene fusion breakpoint classification using gene fusion breakpoint DNA sequences.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: middle embedding with neural-network classifier. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"},{"label":"Inputs","value":"gene fusion breakpoint DNA sequences","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"},{"label":"Assessment","value":"ROC AUC","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"},{"label":"Recorded split or evaluation setting","value":"full test set","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"}],"diagram":{"title":"Reported evaluation outline","steps":["gene fusion breakpoint DNA sequences","Recorded fitting or scoring procedure","Assess ROC AUC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["fusion-breakpoint-foundation-models-2026"],"source_locator":"Table 2, NT / NN (middle) row, ROC AUC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for gene fusion breakpoint DNA sequences have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-f0ed5188dbb6d4","kind":"benchmark","name":"protein-protein binding-site prediction","description":"","status":"needs_review","facets":{"areas":["proteins-complexes"],"tasks":["protein-protein binding-site prediction"]},"source_ids":["protein-binding-sites-2023"],"links":[{"relation":"dataset","target_id":"reported-dataset-f08b1a60aebeeb"}],"attributes":{"entity_level":"task","version":null,"task":"protein-protein binding-site prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests protein-protein binding-site prediction using Dset_448.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Explainable ensemble binding-site predictor using ProtT5 features. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"},{"label":"Inputs","value":"Dset_448","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"},{"label":"Assessment","value":"AUROC","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"}],"diagram":{"title":"Reported evaluation outline","steps":["Dset_448","Recorded fitting or scoring procedure","Assess AUROC"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["protein-binding-sites-2023"],"source_locator":"Table 2, Dset_448 section, ProtT5 row, AUROC column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for Dset_448 have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-f3a12dbc0e0439","kind":"benchmark","name":"Antibody loop structure prediction","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Antibody loop structure prediction"]},"source_ids":["ibex-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-b7204b005bd476"}],"attributes":{"entity_level":"task","version":null,"task":"Antibody loop structure prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Antibody loop structure prediction using ImmuneBuilder antibody test set.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Backbone RMSD after framework alignment; average over antibody test structures.; Backbone RMSD after framework alignment; one seed and one diffusion trajectory. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column; Table 1, Antibodies / Chai-1 row, CDR H3 column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column; Table 1, Antibodies / Chai-1 row, CDR H3 column"},{"label":"Inputs","value":"ImmuneBuilder antibody test set","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column; Table 1, Antibodies / Chai-1 row, CDR H3 column"},{"label":"Assessment","value":"Mean CDR H3 RMSD","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column; Table 1, Antibodies / Chai-1 row, CDR H3 column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column; Table 1, Antibodies / Chai-1 row, CDR H3 column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column; Table 1, Antibodies / Chai-1 row, CDR H3 column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column; Table 1, Antibodies / Chai-1 row, CDR H3 column"}],"diagram":{"title":"Reported evaluation outline","steps":["ImmuneBuilder antibody test set","Recorded fitting or scoring procedure","Assess Mean CDR H3 RMSD"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["ibex-2025"],"source_locator":"Table 1, Antibodies / Ibex row, CDR H3 column; Table 1, Antibodies / Chai-1 row, CDR H3 column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for ImmuneBuilder antibody test set have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-f4b1c9373f0929","kind":"benchmark","name":"Hierarchical metagenomic taxonomy classification","description":"","status":"needs_review","facets":{"areas":["microbes-communities"],"tasks":["Hierarchical metagenomic taxonomy classification"]},"source_ids":["icctax-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-561834dfa1682c"}],"attributes":{"entity_level":"task","version":null,"task":"Hierarchical metagenomic taxonomy classification","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Hierarchical metagenomic taxonomy classification using ICCTax Complete dataset.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Macro average precision at genus rank on Complete dataset. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column; Table 2, Kraken2 row, Genus column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column; Table 2, Kraken2 row, Genus column"},{"label":"Inputs","value":"ICCTax Complete dataset","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column; Table 2, Kraken2 row, Genus column"},{"label":"Assessment","value":"Genus macro AveP","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column; Table 2, Kraken2 row, Genus column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column; Table 2, Kraken2 row, Genus column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column; Table 2, Kraken2 row, Genus column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column; Table 2, Kraken2 row, Genus column"}],"diagram":{"title":"Reported evaluation outline","steps":["ICCTax Complete dataset","Recorded fitting or scoring procedure","Assess Genus macro AveP"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["icctax-2025"],"source_locator":"Table 2, ICCTax row, Genus column; Table 2, Kraken2 row, Genus column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for ICCTax Complete dataset have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-f7142c3b3e0f3c","kind":"benchmark","name":"translation-efficiency prediction","description":"","status":"needs_review","facets":{"areas":["rna-transcriptomes"],"tasks":["translation-efficiency prediction"]},"source_ids":["mrnabert-2025"],"links":[{"relation":"dataset","target_id":"reported-dataset-1744719eef145b"}],"attributes":{"entity_level":"task","version":null,"task":"translation-efficiency prediction","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests translation-efficiency prediction using human ultra-long mRNAs.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: human translation-efficiency regression at 3066-nt input. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"},{"label":"Inputs","value":"human ultra-long mRNAs","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"},{"label":"Assessment","value":"R-squared","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"},{"label":"Recorded split or evaluation setting","value":"paper evaluation","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"}],"diagram":{"title":"Reported evaluation outline","steps":["human ultra-long mRNAs","Recorded fitting or scoring procedure","Assess R-squared"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["mrnabert-2025"],"source_locator":"Table 2, mRNABERT (3066) row, Human R-squared column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for human ultra-long mRNAs have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"reported-task-ff2dec63c5a3dd","kind":"benchmark","name":"Lipid–protein binding pose","description":"","status":"needs_review","facets":{"areas":["molecular-interactions"],"tasks":["Lipid–protein binding pose"]},"source_ids":["lipp-2026"],"links":[{"relation":"dataset","target_id":"reported-dataset-6a44f5946cd7ab"}],"attributes":{"entity_level":"task","version":null,"task":"Lipid–protein binding pose","scope_note":"Paper-specific evaluation task; protocol completeness requires further extraction.","missing_metadata":{"protocol_version":"not_reported_in_legacy_extract","split":"not_reported_in_legacy_extract"},"profile":{"summary":"This paper-specific evaluation tests Lipid–protein binding pose using LiPP lipid–protein complexes.","sections":[{"title":"Evaluation context","body":"The existing paper extraction describes: Top-scoring pose; all-atom lipid RMSD below 2 Å. This description is retained with the exact evaluation records; it is not a new protocol reconstruction.","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column; Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"}],"facts":[{"label":"Record type","value":"Paper-specific task; protocol incompletely extracted","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column; Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"},{"label":"Inputs","value":"LiPP lipid–protein complexes","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column; Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"},{"label":"Assessment","value":"Success rate, ligand all-atom RMSD <2 Å","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column; Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"},{"label":"Recorded split or evaluation setting","value":"Unextracted","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column; Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"}],"strengths":[{"text":"The linked evaluation identifies the paper-specific dataset and assessment rather than treating the task name as a universal benchmark.","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column; Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"}],"limitations":[{"text":"The numerical table check does not establish the full data-processing, fitting or leakage-control protocol.","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column; Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"}],"diagram":{"title":"Reported evaluation outline","steps":["LiPP lipid–protein complexes","Recorded fitting or scoring procedure","Assess Success rate, ligand all-atom RMSD <2 Å"],"caption":"Outline of the existing paper extraction. Split membership, fitting details and scorer implementation remain incompletely reviewed.","source_ids":["lipp-2026"],"source_locator":"Table 2, Chai-1 row, LiPP (N=331) % Success Rate column; Table 2, DiffDock-L row, LiPP (N=331) % Success Rate column"},"coverage":"limited","gaps":["Dataset release/accession and complete split manifest for LiPP lipid–protein complexes have not been verified in this profile.","Allowed inputs, model selection and evaluator implementation require methods-level review."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Catalogue extraction inspected; protocol claims remain limited to the cited evidence. Missing details are not presumed absent from the original paper."}}}} {"id":"rewire-evaluation-baseline-kmer-position-v2","kind":"evaluation","name":"Corrected k-mer / position baseline on MFASS v2","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"model","target_id":"rewire-model-baseline-kmer-position-v2"},{"relation":"benchmark","target_id":"rewire-mfass-v2"},{"relation":"dataset","target_id":"rewire-mfass-v2-dataset"}],"attributes":{"origin":"rewire_run","protocol":"Assay-oriented 21 bp k-mer window, exon position, allele identity and conservation features; gradient-boosted trees trained on the MFASS training split.","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","run_url":"/benchmarks/runs/mfass-v2/","comparison":{"protocol_id":"mfass-v2","dataset_version":"bee9133b83f3aedaf2bbb9013f1875515845607e","split":"split-v2.tsv","population":"8324/8324","inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"coverage":"8324/8324","missing_metadata":{"paired_comparison":"Refer to paired bootstrap artifacts; marginal scores are not paired comparisons."}}} {"id":"rewire-evaluation-dnabert2-117m-frozen-pair-logreg","kind":"evaluation","name":"DNABERT-2 117M · frozen pair embeddings on MFASS v2","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"model","target_id":"rewire-model-dnabert2-117m-frozen-pair-logreg"},{"relation":"benchmark","target_id":"rewire-mfass-v2"},{"relation":"dataset","target_id":"rewire-mfass-v2-dataset"}],"attributes":{"origin":"rewire_run","protocol":"Masked mean of frozen last hidden states for 170 bp reference and mutant sequences; concatenate reference and mutant-minus-reference embeddings; fixed balanced L2 logistic head trained only on the MFASS training split.","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","run_url":"/benchmarks/runs/mfass-v2/","comparison":{"protocol_id":"mfass-v2","dataset_version":"bee9133b83f3aedaf2bbb9013f1875515845607e","split":"split-v2.tsv","population":"8324/8324","inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"coverage":"8324/8324","missing_metadata":{"paired_comparison":"Refer to paired bootstrap artifacts; marginal scores are not paired comparisons."}}} {"id":"rewire-evaluation-pangolin-maskfalse","kind":"evaluation","name":"Pangolin · mask=False on MFASS v2","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"model","target_id":"rewire-model-pangolin-maskfalse"},{"relation":"benchmark","target_id":"rewire-mfass-v2"},{"relation":"dataset","target_id":"rewire-mfass-v2-dataset"}],"attributes":{"origin":"rewire_run","protocol":"Unchanged specialist run in genomic context with GENCODE v44; zero-shot on MFASS assay labels. Point metrics use the scored subset.","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","run_url":"/benchmarks/runs/mfass-v2/","comparison":{"protocol_id":"mfass-v2","dataset_version":"bee9133b83f3aedaf2bbb9013f1875515845607e","split":"split-v2.tsv","population":"8301/8324","inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"coverage":"8301/8324","missing_metadata":{"paired_comparison":"Refer to paired bootstrap artifacts; marginal scores are not paired comparisons."}}} {"id":"rewire-evaluation-spliceai-1-3-1","kind":"evaluation","name":"SpliceAI 1.3.1 on MFASS v2","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"model","target_id":"rewire-model-spliceai-1-3-1"},{"relation":"benchmark","target_id":"rewire-mfass-v2"},{"relation":"dataset","target_id":"rewire-mfass-v2-dataset"}],"attributes":{"origin":"rewire_run","protocol":"Unchanged specialist run in genomic context with bundled annotation; zero-shot on MFASS assay labels. Point metrics use the scored subset.","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","run_url":"/benchmarks/runs/mfass-v2/","comparison":{"protocol_id":"mfass-v2","dataset_version":"bee9133b83f3aedaf2bbb9013f1875515845607e","split":"split-v2.tsv","population":"8194/8324","inputs":null,"adaptation":null,"metric_implementation":null,"aggregation":null,"budget":null},"coverage":"8194/8324","missing_metadata":{"paired_comparison":"Refer to paired bootstrap artifacts; marginal scores are not paired comparisons."}}} {"id":"rewire-mfass-v1","kind":"benchmark","name":"MFASS v1 (superseded)","description":"The mfass-v1 baseline used a mis-centred k-mer window for 7,770 assay variants whose raw sequence was reverse-complemented. mfass-v2 validates assay-oriented reference and mutant pairs and rebuilds the baseline; mfass-v1 remains a historical record.","status":"superseded","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluates_task","target_id":"catalog-task-mfass-splice"}],"attributes":{"entity_level":"protocol","version":"v1","task":"Splice-variant prioritisation","scope_note":"The mfass-v1 baseline used a mis-centred k-mer window for 7,770 assay variants whose raw sequence was reverse-complemented. mfass-v2 validates assay-oriented reference and mutant pairs and rebuilds the baseline; mfass-v1 remains a historical record.","missing_metadata":{},"profile":{"summary":"MFASS v1 is an archived protocol with an invalid baseline-window implementation.","sections":[{"title":"Procedure","body":"The original sequence field was combined with assay-oriented coordinates. For 7,770 eligible variants, the field was reverse complemented and the local feature window was centred incorrectly. Use v2 for current baseline comparisons; retain v1 only to audit the correction.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"facts":[{"label":"Record type","value":"Superseded protocol","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},{"label":"Inputs","value":"Historical MFASS variant features","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},{"label":"Assessment","value":"Archived results, withdrawn as current baseline evidence","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"strengths":[{"text":"Preserving the original artifacts makes the scientific correction inspectable.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"limitations":[{"text":"Do not use v1 baseline comparisons or split-cost claims to select a current method.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"diagram":{"title":"Correction history","steps":["Original raw-sequence windows","Orientation mismatch identified","Assay-oriented pairs validated","Use corrected v2 comparisons"],"caption":"Historical correction path. Do not use the superseded v1 baseline for current model selection.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}}} {"id":"rewire-mfass-v2","kind":"benchmark","name":"MFASS v2","description":"","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"supersedes","target_id":"rewire-mfass-v1"},{"relation":"evaluates_task","target_id":"catalog-task-mfass-splice"}],"attributes":{"entity_level":"protocol","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","task":"Splice-variant prioritisation","scope_note":"The mfass-v1 baseline used a mis-centred k-mer window for 7,770 assay variants whose raw sequence was reverse-complemented. mfass-v2 validates assay-oriented reference and mutant pairs and rebuilds the baseline; mfass-v1 remains a historical record.","missing_metadata":{},"profile":{"summary":"MFASS v2 ranks splice-disrupting variants using corrected, assay-oriented sequence inputs.","sections":[{"title":"Procedure","body":"Validate reference and mutant sequences, apply the predeclared gene-and-exon connected-component split, then rank held-out variants. Report top-100 precision, average precision and AUROC; paired comparisons use common scored variants and group-aware uncertainty.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"facts":[{"label":"Record type","value":"Pinned rewire protocol","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},{"label":"Inputs","value":"MFASS functional assay labels; assay-oriented variant pairs or specialist genomic inputs","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},{"label":"Assessment","value":"Variant ranking on the held-out assay set","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},{"label":"Eligible cohort / train / test","value":"27,733 / 19,409 / 8,324 variants","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/results/baseline-kmer-position-v2.json; README: Cohort reconciliation and Reproduce v2"},{"label":"Corrected sequence handling","value":"Reference natural_seq and mutant original_seq are validated at rel_position; 7,770 eligible raw sequence fields were reverse complemented.","source_ids":["rewire-mfass-v2-source"],"source_locator":"README: Correction and Cohort reconciliation"},{"label":"Baseline","value":"Gradient-boosted trees using a correctly centred 21 bp k-mer window, exon position, alleles and conservation.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/results/baseline-kmer-position-v2.json"}],"strengths":[{"text":"A fixed review capacity connects ranking quality to a limited follow-up budget.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"limitations":[{"text":"The reporter measures exon recognition in an artificial construct, not clinical pathogenicity. Methods have different input context and scoring coverage.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"}],"diagram":{"title":"Procedure overview","steps":["Validate assay-oriented pairs","Use pinned grouped split","Score held-out variants","Assess ranking and coverage"],"caption":"Conceptual overview, not an executable specification.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md at bee9133b83f3aedaf2bbb9013f1875515845607e: Correction; Dataset; Cohort reconciliation; Split; results JSON in benchmarks/mfass/results/"},"coverage":"reviewed","gaps":[],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Primary-source description checked by an automated research assistant. This is profile review, not independent execution or numerical reproduction."}}}} {"id":"rewire-mfass-v2-dataset","kind":"dataset","name":"MFASS v2 eligible assay cohort","description":"","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[],"attributes":{"version":"bee9133b83f3aedaf2bbb9013f1875515845607e","split":"split-v2.tsv","cohort_variants":27733,"train_variants":19409,"test_variants":8324,"test_positives":315,"independent_test_groups":463,"missing_metadata":{}}} {"id":"rewire-mfass-v2-source","kind":"source","name":"MFASS v2 pinned rewire artifacts","description":"","status":"source_checked","facets":{},"source_ids":[],"links":[],"attributes":{"url":"https://github.com/timini/rewire-benchmarks/tree/bee9133b83f3aedaf2bbb9013f1875515845607e","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","retrieved_at":"2026-09-16T10:50:02Z","artifact_sha256":"a3af693afc070b39b334e359beed7f9affd76f7456a4f9264e2d1e9dab26111d"}} {"id":"rewire-model-baseline-kmer-position-v2","kind":"model","name":"Corrected k-mer / position baseline","description":"","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[],"attributes":{"entity_level":"method","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","reported_name":"Corrected k-mer / position baseline","missing_metadata":{},"profile":{"summary":"The corrected MFASS baseline combines local sequence composition, allele identity, exon position and conservation in gradient-boosted trees.","sections":[{"title":"Evaluated procedure","body":"Features use an assay-oriented 21-base window centred on the validated variant position. The model is trained on the fixed MFASS training split, then scores the held-out variants. The corrected orientation is part of the method identity.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/src/mfass/run_baseline.py, featurise; results/baseline-kmer-position-v2.json at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"facts":[{"label":"Sequence features","value":"3-mer composition in a 21-base assay-oriented window","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/src/mfass/run_baseline.py, featurise; results/baseline-kmer-position-v2.json at bee9133b83f3aedaf2bbb9013f1875515845607e"},{"label":"Other inputs","value":"Exon-boundary distances, allele identity, phyloP and phastCons","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/src/mfass/run_baseline.py, featurise; results/baseline-kmer-position-v2.json at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"strengths":[{"text":"Provides an interpretable feature-based reference for asking whether a more complex model adds value.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/src/mfass/run_baseline.py, featurise; results/baseline-kmer-position-v2.json at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"limitations":[{"text":"It uses assay-specific labels and engineered annotation features. It is not a zero-shot baseline with the same inputs as every specialist.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/src/mfass/run_baseline.py, featurise; results/baseline-kmer-position-v2.json at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"diagram":{"title":"Evaluated pipeline","steps":["Validated assay sequence","21-base window and features","Training split","Gradient-boosted trees","Held-out ranking"],"caption":"Schematic of the pinned MFASS configuration; this does not generalise to every member of the model family.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/src/mfass/run_baseline.py, featurise; results/baseline-kmer-position-v2.json at bee9133b83f3aedaf2bbb9013f1875515845607e"},"coverage":"reviewed","gaps":["Archived v1 outputs used mis-centred windows and must not substitute for this v2 configuration."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected the pinned runner README and result configuration with git show. No models were run; existing result and reproduction statuses are unchanged."}}}} {"id":"rewire-model-dnabert2-117m-frozen-pair-logreg","kind":"model","name":"DNABERT-2 117M · frozen pair embeddings","description":"","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"uses_model","target_id":"discovery-model-dnabert-2"}],"attributes":{"entity_level":"method","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","reported_name":"DNABERT-2 117M · frozen pair embeddings","missing_metadata":{},"profile":{"summary":"This MFASS pipeline uses a frozen DNABERT-2 encoder with an assay-specific logistic-regression head. It is distinct from the base encoder and from end-to-end fine-tuning.","sections":[{"title":"Evaluated procedure","body":"Validated 170-base reference and mutant sequences are embedded separately. Attention-mask means of final hidden states are combined as reference plus mutant-minus-reference. Standardisation and balanced L2 logistic regression are fitted only on the training arm.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, DNABERT-2 protocol; results/dnabert2-117m-frozen-pair-logreg.json, config at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"facts":[{"label":"Encoder revision","value":"b5ae377faa374ee160eec1c27b8494436cc94451","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, DNABERT-2 protocol; results/dnabert2-117m-frozen-pair-logreg.json, config at bee9133b83f3aedaf2bbb9013f1875515845607e"},{"label":"Head","value":"StandardScaler + balanced L2 logistic regression, C=0.1","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, DNABERT-2 protocol; results/dnabert2-117m-frozen-pair-logreg.json, config at bee9133b83f3aedaf2bbb9013f1875515845607e"},{"label":"Context","value":"170 bases for each reference and mutant sequence","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, DNABERT-2 protocol; results/dnabert2-117m-frozen-pair-logreg.json, config at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"strengths":[{"text":"Separates the utility of a frozen representation from encoder fine-tuning and pins the checkpoint and head settings.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, DNABERT-2 protocol; results/dnabert2-117m-frozen-pair-logreg.json, config at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"limitations":[{"text":"The result tests this short-context supervised pipeline. It does not establish how all DNABERT-2 adaptations or all foundation models perform.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, DNABERT-2 protocol; results/dnabert2-117m-frozen-pair-logreg.json, config at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"diagram":{"title":"Evaluated pipeline","steps":["Assay-oriented sequence pair","Frozen DNABERT-2","Masked mean embeddings","Reference plus difference","Trained logistic head"],"caption":"Schematic of the pinned MFASS configuration; this does not generalise to every member of the model family.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, DNABERT-2 protocol; results/dnabert2-117m-frozen-pair-logreg.json, config at bee9133b83f3aedaf2bbb9013f1875515845607e"},"coverage":"reviewed","gaps":["Overlap between the assay sequences and the encoder’s pretraining corpus has not been checked."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected the pinned runner README and result configuration with git show. No models were run; existing result and reproduction statuses are unchanged."}}}} {"id":"rewire-model-pangolin-maskfalse","kind":"model","name":"Pangolin · mask=False","description":"","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"uses_model","target_id":"discovery-model-pangolin"}],"attributes":{"entity_level":"method","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","reported_name":"Pangolin · mask=False","missing_metadata":{},"profile":{"summary":"This MFASS evaluation uses Pangolin with masking disabled, scoring variants in genomic context.","sections":[{"title":"Evaluated procedure","body":"The runner uses the specified reference and GENCODE annotation to predict changes in splice strength. The mask=False choice retains changes that annotation-based masking would remove. These specialist predictions were retained unchanged in the corrected v2 comparison.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/pangolin-maskFalse.json, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"facts":[{"label":"Mask","value":"False","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/pangolin-maskFalse.json, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"},{"label":"Assay-label adaptation","value":"Zero-shot on MFASS labels","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/pangolin-maskFalse.json, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"strengths":[{"text":"Tests a pretrained splice specialist without learning from MFASS assay labels.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/pangolin-maskFalse.json, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"limitations":[{"text":"Results depend on annotation, masking and scored coverage. This configuration is not interchangeable with Pangolin’s masked default.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/pangolin-maskFalse.json, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"diagram":{"title":"Evaluated pipeline","steps":["Variant and genomic context","Pangolin predictor","Splice-strength changes","mask=False aggregation","Variant ranking"],"caption":"Schematic of the pinned MFASS configuration; this does not generalise to every member of the model family.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/pangolin-maskFalse.json, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"},"coverage":"reviewed","gaps":["Potential overlap with specialist pretraining sequences is not resolved by this comparison."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected the pinned runner README and result configuration with git show. No models were run; existing result and reproduction statuses are unchanged."}}}} {"id":"rewire-model-spliceai-1-3-1","kind":"model","name":"SpliceAI 1.3.1","description":"","status":"source_checked","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"uses_model","target_id":"discovery-model-spliceai"}],"attributes":{"entity_level":"method","version":"bee9133b83f3aedaf2bbb9013f1875515845607e","reported_name":"SpliceAI 1.3.1","missing_metadata":{},"profile":{"summary":"This MFASS evaluation uses the official SpliceAI 1.3.1 five-model ensemble in genomic context.","sections":[{"title":"Evaluated procedure","body":"The runner scores variants against GRCh38 with bundled annotations and unmasked outputs. It ranks variants by the largest acceptor/donor gain or loss delta score. Its specialist predictions were retained unchanged for the corrected v2 comparison.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/spliceai-1.3.1.json, description, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"facts":[{"label":"Version","value":"SpliceAI 1.3.1, bundled five-model ensemble","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/spliceai-1.3.1.json, description, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"},{"label":"Score","value":"Maximum of DS_AG, DS_AL, DS_DG and DS_DL","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/spliceai-1.3.1.json, description, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"},{"label":"Mask","value":"M=0","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/spliceai-1.3.1.json, description, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"strengths":[{"text":"Uses a pretrained splice specialist without fitting to MFASS labels.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/spliceai-1.3.1.json, description, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"limitations":[{"text":"The genomic context differs from the assay construct. Missing predictions and possible exon overlap with specialist training data must be considered.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/spliceai-1.3.1.json, description, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"}],"diagram":{"title":"Evaluated pipeline","steps":["Variant in GRCh38","SpliceAI ensemble","Acceptor / donor delta scores","Maximum delta","Variant ranking"],"caption":"Schematic of the pinned MFASS configuration; this does not generalise to every member of the model family.","source_ids":["rewire-mfass-v2-source"],"source_locator":"benchmarks/mfass/README.md, MFASS-v2 results; results/spliceai-1.3.1.json, description, config and notes at bee9133b83f3aedaf2bbb9013f1875515845607e"},"coverage":"reviewed","gaps":["Assayed-exon overlap with the specialist training transcripts is unchecked."],"review":{"method":"automated_source_review","date":"2026-09-16","note":"Inspected the pinned runner README and result configuration with git show. No models were run; existing result and reproduction statuses are unchanged."}}}} {"id":"rewire-result-baseline-kmer-position-v2-auroc","kind":"result","name":"Corrected k-mer / position baseline · auroc","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-baseline-kmer-position-v2"}],"attributes":{"printed_value":"0.7779498064677238","numeric_value":"0.7779498064677238","metric":"auroc","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/baseline-kmer-position-v2.json :: auroc","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8324/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-baseline-kmer-position-v2-average-precision","kind":"result","name":"Corrected k-mer / position baseline · average_precision","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-baseline-kmer-position-v2"}],"attributes":{"printed_value":"0.28641674595892375","numeric_value":"0.28641674595892375","metric":"average_precision","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/baseline-kmer-position-v2.json :: average_precision","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8324/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-baseline-kmer-position-v2-precision-at-100","kind":"result","name":"Corrected k-mer / position baseline · precision_at_100","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-baseline-kmer-position-v2"}],"attributes":{"printed_value":"0.61","numeric_value":"0.61","metric":"precision_at_100","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/baseline-kmer-position-v2.json :: precision_at_100","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8324/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-dnabert2-117m-frozen-pair-logreg-auroc","kind":"result","name":"DNABERT-2 117M · frozen pair embeddings · auroc","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-dnabert2-117m-frozen-pair-logreg"}],"attributes":{"printed_value":"0.5500324040216661","numeric_value":"0.5500324040216661","metric":"auroc","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.json :: auroc","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8324/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-dnabert2-117m-frozen-pair-logreg-average-precision","kind":"result","name":"DNABERT-2 117M · frozen pair embeddings · average_precision","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-dnabert2-117m-frozen-pair-logreg"}],"attributes":{"printed_value":"0.04508654312652131","numeric_value":"0.04508654312652131","metric":"average_precision","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.json :: average_precision","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8324/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-dnabert2-117m-frozen-pair-logreg-precision-at-100","kind":"result","name":"DNABERT-2 117M · frozen pair embeddings · precision_at_100","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-dnabert2-117m-frozen-pair-logreg"}],"attributes":{"printed_value":"0.03","numeric_value":"0.03","metric":"precision_at_100","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/dnabert2-117m-frozen-pair-logreg.json :: precision_at_100","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8324/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-pangolin-maskfalse-auroc","kind":"result","name":"Pangolin · mask=False · auroc","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-pangolin-maskfalse"}],"attributes":{"printed_value":"0.8756851300560864","numeric_value":"0.8756851300560864","metric":"auroc","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/pangolin-maskFalse.json :: auroc","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8301/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-pangolin-maskfalse-average-precision","kind":"result","name":"Pangolin · mask=False · average_precision","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-pangolin-maskfalse"}],"attributes":{"printed_value":"0.3887617543064248","numeric_value":"0.3887617543064248","metric":"average_precision","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/pangolin-maskFalse.json :: average_precision","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8301/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-pangolin-maskfalse-precision-at-100","kind":"result","name":"Pangolin · mask=False · precision_at_100","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-pangolin-maskfalse"}],"attributes":{"printed_value":"0.65","numeric_value":"0.65","metric":"precision_at_100","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/pangolin-maskFalse.json :: precision_at_100","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8301/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-spliceai-1-3-1-auroc","kind":"result","name":"SpliceAI 1.3.1 · auroc","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-spliceai-1-3-1"}],"attributes":{"printed_value":"0.8055241740253153","numeric_value":"0.8055241740253153","metric":"auroc","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/spliceai-1.3.1.json :: auroc","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8194/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-spliceai-1-3-1-average-precision","kind":"result","name":"SpliceAI 1.3.1 · average_precision","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-spliceai-1-3-1"}],"attributes":{"printed_value":"0.2986855472760137","numeric_value":"0.2986855472760137","metric":"average_precision","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/spliceai-1.3.1.json :: average_precision","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8194/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rewire-result-spliceai-1-3-1-precision-at-100","kind":"result","name":"SpliceAI 1.3.1 · precision_at_100","description":"","status":"reproduced","facets":{"areas":["dna-genomes"]},"source_ids":["rewire-mfass-v2-source"],"links":[{"relation":"evaluation","target_id":"rewire-evaluation-spliceai-1-3-1"}],"attributes":{"printed_value":"0.64","numeric_value":"0.64","metric":"precision_at_100","metric_direction":"higher","unit":"fraction","uncertainty":null,"source_locator":"benchmarks/mfass/results/spliceai-1.3.1.json :: precision_at_100","review":{"method":"existing_run_import","reviewer":"rewire documented MFASS v2 run","reviewed_at":"2026-09-16T10:50:02Z","notes":"Imported existing documented own-run record, not a newly executed reproduction."},"coverage":"8194/8324","missing_metadata":{"uncertainty":"Point estimate; see paired-comparison artifacts."}}} {"id":"rlsite-rna-binding-2025","kind":"source","name":"RNA language model and graph attention network for RNA and small molecule binding sites prediction","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12417085/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf447","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"a50f344e253162ae43f51d7120cfb35a1d0f6114fd8176d760aceb6d05fd95bd","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12417085/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558222+00:00","legacy_paper":{"id":"rlsite-rna-binding-2025","title":"RNA language model and graph attention network for RNA and small molecule binding sites prediction","year":2025,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12417085/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioinformatics/btaf447","notes":"Numeric result checked against Table 1. in primary full-text XML; journal/source: Bioinformatics."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"rnaret-2026","kind":"source","name":"Retentive Network promotes efficient RNA language modeling of long sequences","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13111708/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1038/s42003-026-09757-x","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"e970e7322e07fb3c9d12efd315691cc5de5575a3f2616f4b788614c8c706dd0b","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13111708/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558224+00:00","legacy_paper":{"id":"rnaret-2026","title":"Retentive Network promotes efficient RNA language modeling of long sequences","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13111708/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1038/s42003-026-09757-x","notes":"Numeric result checked against Table 1 in primary full-text XML; journal/source: Communications Biology."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"scalr-2025","kind":"source","name":"scaLR: a low-resource deep neural network-based platform for single cell analysis and biomarker discovery","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12121358/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1093/bib/bbaf243","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"829afab6a4e30997c608745d3eb280105b8c5c4e601020ffdc55c866144527ca","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12121358/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.403844+00:00","legacy_paper":{"id":"scalr-2025","title":"scaLR: a low-resource deep neural network-based platform for single cell analysis and biomarker discovery","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12121358/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Briefings in Bioinformatics; PMC ID: PMC12121358.","doi":"10.1093/bib/bbaf243"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"scatac-llmda-2026","kind":"source","name":"Cell type annotation for scATAC-seq via DNA large language model and graph domain adaptation","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13132462/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1371/journal.pcbi.1014226","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"f1cdc7d54c6b2d491e4a74a44a1188a3679c555262c988e95a1c0de4614fe3cb","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13132462/fullTextXML","artifact_retrieved_at":"2026-09-16T10:44:03.395850+00:00","legacy_paper":{"id":"scatac-llmda-2026","title":"Cell type annotation for scATAC-seq via DNA large language model and graph domain adaptation","year":2026,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13132462/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: PLOS Computational Biology; PMC ID: PMC13132462.","doi":"10.1371/journal.pcbi.1014226"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"scelmo-2025","kind":"source","name":"scELMo: Embeddings from Language Models are Good Learners for Single-cell Data Analysis","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12393277/","version":"preprint archived 2025-08-23","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2023.12.07.569910","publication_status":"preprint","year":2025,"artifact_sha256":"ef75f0d63a567f5e9d7132fd847f44838a82a9741ae55323437e1d1812d86316","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12393277/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.537541+00:00","legacy_paper":{"id":"scelmo-2025","title":"scELMo: Embeddings from Language Models are Good Learners for Single-cell Data Analysis","year":2025,"publication_status":"preprint","version":"preprint archived 2025-08-23","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12393277/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC12393277.","doi":"10.1101/2023.12.07.569910"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"scregnet-2025","kind":"source","name":"Prediction of Gene Regulatory Connections with Joint Single-Cell Foundation Models and Graph-Based Learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838224/","version":"PMC11838224.2","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2024.12.16.628715","publication_status":"preprint","year":2025,"artifact_sha256":"65b3272d47bb9c4ee1e7a965169bef63add9dbeb31508d4076e5145b761af4ec","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11838224/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.541287+00:00","legacy_paper":{"id":"scregnet-2025","title":"Prediction of Gene Regulatory Connections with Joint Single-Cell Foundation Models and Graph-Based Learning","year":2025,"publication_status":"preprint","version":"PMC11838224.2","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11838224/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC11838224.","doi":"10.1101/2024.12.16.628715"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"scxdr-2026","kind":"source","name":"scXDR: drug response prediction across single-cell datasets via heterogeneous network transfer learning","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12859067/","version":"PMC archival version PMC12859067.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1038/s42003-025-09418-5","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"47b5925e9887d87fc8288d29288802b1d67d54f064d913151df92171f7c68d33","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12859067/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:50.056Z","legacy_paper":{"id":"scxdr-2026","title":"scXDR: drug response prediction across single-cell datasets via heterogeneous network transfer learning","year":2026,"publication_status":"peer_reviewed","version":"PMC archival version PMC12859067.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12859067/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Communications Biology; PMC ID: PMC12859067.","doi":"10.1038/s42003-025-09418-5"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"single-cell-aging-probes-2026","kind":"source","name":"Inflammation-linked aging signals in frozen single-cell foundation models: donor-aware detection and robustness testing","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13407579/","version":"PMC archival version PMC13407579.1","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1007/s10522-026-10471-8","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"d6cfb13933ceefed630954f804e7dc979b747bc4aec4c8c5518232f25772736a","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13407579/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:44.421Z","legacy_paper":{"id":"single-cell-aging-probes-2026","title":"Inflammation-linked aging signals in frozen single-cell foundation models: donor-aware detection and robustness testing","year":2026,"publication_status":"peer_reviewed","version":"PMC archival version PMC13407579.1","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13407579/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: Biogerontology; PMC ID: PMC13407579.","doi":"10.1007/s10522-026-10471-8"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"single-cell-peft-2024","kind":"source","name":"Parameter-Efficient Fine-Tuning Enhances Adaptation of Single Cell Large Language Model for Cell Type Identification","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10862733/","version":"preprint archived 2024-01-30","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1101/2024.01.27.577455","publication_status":"preprint","year":2024,"artifact_sha256":"77a4a859010259eadf2187465db6ab385efa4927a5eadb95c1e01991044c283f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC10862733/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.530269+00:00","legacy_paper":{"id":"single-cell-peft-2024","title":"Parameter-Efficient Fine-Tuning Enhances Adaptation of Single Cell Large Language Model for Cell Type Identification","year":2024,"publication_status":"preprint","version":"preprint archived 2024-01-30","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC10862733/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: bioRxiv; PMC ID: PMC10862733.","doi":"10.1101/2024.01.27.577455"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"single-cell-residual-geometry-2026","kind":"source","name":"Residual-stream geometry of single-cell foundation models carries incremental gene-regulatory signal across tissues","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["cells-tissues"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13418759/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1186/s12859-026-06538-5","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"77546faec51cfb5c78b73c5943f940c4e0097d130b1ddde4ab8287b507b36df6","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13418759/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"single-cell-residual-geometry-2026","title":"Residual-stream geometry of single-cell foundation models carries incremental gene-regulatory signal across tissues","year":2026,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13418759/","primary_domain":"cells-tissues","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: BMC Bioinformatics; PMC ID: PMC13418759. Score reflects scGPT plus paper geometry features, not raw scGPT.","doi":"10.1186/s12859-026-06538-5"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"spin-protein-function-2026","kind":"source","name":"Scaling the profile of life by function with SPIN","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12970593/","version":"journal full text in PMC","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1093/bioadv/vbag064","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"9701843e93bf7fa3ead71e19693fb07d483f1022379871adfb04486783722a9d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12970593/fullTextXML","artifact_retrieved_at":"2026-09-16T10:38:57.558225+00:00","legacy_paper":{"id":"spin-protein-function-2026","title":"Scaling the profile of life by function with SPIN","year":2026,"publication_status":"peer_reviewed","version":"journal full text in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12970593/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1093/bioadv/vbag064","notes":"Numeric result checked against Table 1 in primary full-text XML; journal/source: Bioinformatics Advances."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"attributes":{"artifact_sha256":"045aa63715acf615327d45ff42a897e77fa6d4b88084d2ad4e2ea836eb5fb48a","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:32.390263+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/aertslab/GENIE3/blob/54bc15636322e8773357de6e0b6683c6bc802825/README.md","version":"54bc15636322e8773357de6e0b6683c6bc802825"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-aertslab-genie3","kind":"source","links":[],"name":"aertslab/GENIE3 official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"f2befd99acf59576a22b8a44abd2345e8ed7304cf470609f27d311e08ed3f066","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:28.363182+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/aertslab/GRNBoost/blob/26c836b3dcbb85852d3c6f4b8340e8655434da02/README.md","version":"26c836b3dcbb85852d3c6f4b8340e8655434da02"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-aertslab-grnboost","kind":"source","links":[],"name":"aertslab/GRNBoost official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"4aa29fa1d4333a72013fd2c60545218f0015b85b95b76878cbdb88962ff4e55d","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:22.733246+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/altoslabs/perturbench/blob/c84038bc1ea409aa54f3832cfa6f34f5059adf0c/README.md","version":"c84038bc1ea409aa54f3832cfa6f34f5059adf0c"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-altoslabs-perturbench","kind":"source","links":[],"name":"altoslabs/perturbench official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"50a634fe3236b6272b70928ac41bdebd4a230467b151f59597742ccd56ac8909","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:33.087504+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/aqlaboratory/genie3/blob/d77ae5ac04212ff1e8b29b585859a3244c614804/README.md","version":"d77ae5ac04212ff1e8b29b585859a3244c614804"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-aqlaboratory-genie3","kind":"source","links":[],"name":"aqlaboratory/genie3 official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"985b6dc09144ea14378dd2c543b288e8d2a05cb342a77cbc56b7a39c5170f638","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:25.490370+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/aqlaboratory/openfold/blob/be2ec1841f16c966c65ae0e7599ebbadc725757d/README.md","version":"be2ec1841f16c966c65ae0e7599ebbadc725757d"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-aqlaboratory-openfold","kind":"source","links":[],"name":"aqlaboratory/openfold official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"58787c8ef5cb4fba4c04322a4ceb9f174e2233ec22d4193622fb6bc67d651d89","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:24.685703+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/ArcInstitute/evo2/blob/53f195997257c56c00e5ef8d33a54f5baad143a6/README.md","version":"53f195997257c56c00e5ef8d33a54f5baad143a6"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-arcinstitute-evo2","kind":"source","links":[],"name":"ArcInstitute/evo2 official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"568c0b4f9d93374f7ebc7467c226fdde0050abfd60147bce0b316163bf4199d6","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:32.302509+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/ArcInstitute/state/blob/9bbfe78a434a55205e4de834e1ea99f85f7a3add/README.md","version":"9bbfe78a434a55205e4de834e1ea99f85f7a3add"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-arcinstitute-state","kind":"source","links":[],"name":"ArcInstitute/state official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"5a089ca429ed2a314e257fddacd70251c863e62a3547b793c38f56785861cac1","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:24.477301+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/Benchmarking-Initiative/Benchmark-Models-PEtab/blob/ddaa86d13f708926c57ec8918ce75a6b50e2e562/README.md","version":"ddaa86d13f708926c57ec8918ce75a6b50e2e562"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-benchmarking-initiative-benchmark-models-petab","kind":"source","links":[],"name":"Benchmarking-Initiative/Benchmark-Models-PEtab official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"96260519b594ae22cba9f28d1f64622de001f2abf11d406c9da572bfaf145727","licence":null,"locator":"readme.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:28.315898+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/biobakery/humann/blob/e07b3a34d0b94c09a8ac5d28ff95009611178be2/readme.md","version":"e07b3a34d0b94c09a8ac5d28ff95009611178be2"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-biobakery-humann","kind":"source","links":[],"name":"biobakery/humann official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"ce491bb2d686145e0773c685d0d02e8a5fabc7daaea60eef07cf54298561fba7","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:27.980580+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/biobakery/MetaPhlAn/blob/424f3e6e30618266404353e1083c6405a9f02f48/README.md","version":"424f3e6e30618266404353e1083c6405a9f02f48"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-biobakery-metaphlan","kind":"source","links":[],"name":"biobakery/MetaPhlAn official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"2736a9546e94b367e1bb3d1052e22460bb2188229d432b71eb9b01fe6b2a9b1a","licence":null,"locator":"readme.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:21.987781+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/biomap-research/PFMBench/blob/53758ffcbdf1d79b5d125383e4dd52d6fd59d2a1/readme.md","version":"53758ffcbdf1d79b5d125383e4dd52d6fd59d2a1"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-biomap-research-pfmbench","kind":"source","links":[],"name":"biomap-research/PFMBench official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"4d907a6723f3f56b14b13e35eeb0833ddf9a2f3512a1fafd07d79259105b264c","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:27.678733+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/BojarLab/glycowork/blob/3d63f1ec25c850da3cde4d25cb602d50b6b5732b/README.md","version":"3d63f1ec25c850da3cde4d25cb602d50b6b5732b"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-bojarlab-glycowork","kind":"source","links":[],"name":"BojarLab/glycowork official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"b0503e8ca789f19f1fc2350c5aaf57b1b323bbae43b354655231b5f4a1586c83","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:26.283123+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/bowang-lab/scGPT/blob/cebd6fae655b9c585a4807daa3ac31bb764f06b4/README.md","version":"cebd6fae655b9c585a4807daa3ac31bb764f06b4"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-bowang-lab-scgpt","kind":"source","links":[],"name":"bowang-lab/scGPT official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"1fbb077b23329940ae6ee1dbcc578831765cc6a3c21e61c830d189a1b4be1fa3","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:55.874128+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://biofunctionprediction.org/cafa/","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-cafa","kind":"source","links":[],"name":"cafa official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"b77d93b4ff352874ecd21e074357b0bb7f4f09e7bfd53d8f7f152d05063c30c1","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:27.117271+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/calico/basenji/blob/06ce5d387e20b47184d05433b3983163c5f923cd/README.md","version":"06ce5d387e20b47184d05433b3983163c5f923cd"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-calico-basenji","kind":"source","links":[],"name":"calico/basenji official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"17825bf33280f40b596a104c547b57fae5ee5c07f8d60b396d0f4780d47ef9a5","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.339676+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://cami-challenge.org/","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-cami","kind":"source","links":[],"name":"cami official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"5908389c2f1f5797ffb73d45776684ff0b34c0fc61a69a2b5b9dbd88df58dcec","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:23.152881+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/CAMI-challenge/AMBER/blob/f8b3a601043d13fc4227d5691c13561eb4490e50/README.md","version":"f8b3a601043d13fc4227d5691c13561eb4490e50"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-cami-challenge-amber","kind":"source","links":[],"name":"CAMI-challenge/AMBER official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"ec5309bf16daab6c9a0adb393b191a3e716b9c627834cf3e08655c250a63da61","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:22.965882+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/CAMI-challenge/CAMISIM/blob/7ce6013c6d5a0fac8ba8a80e52a03560d3546fa6/README.md","version":"7ce6013c6d5a0fac8ba8a80e52a03560d3546fa6"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-cami-challenge-camisim","kind":"source","links":[],"name":"CAMI-challenge/CAMISIM official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"d123b202234d3ed520911982dcf88e2fcfd888c83e65378effb2d2e269d0abb1","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:23.429372+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/CAMI-challenge/OPAL/blob/98120c326eef08e391899e4bd3a362e0e6558b4a/README.md","version":"98120c326eef08e391899e4bd3a362e0e6558b4a"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-cami-challenge-opal","kind":"source","links":[],"name":"CAMI-challenge/OPAL official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"5e22606895cf0c30565ed4456bc680ca97c10c8c5d38054f82a6e4677c8d8c24","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.812767+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://www.capri-docking.org/","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-capri","kind":"source","links":[],"name":"capri official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"de391b68636462ddb78d0659d8784128909d23e9c015832cca94d883f404a3f7","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.981498+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://predictioncenter.org/","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-casp","kind":"source","links":[],"name":"casp official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"4f1728521ab79de1c33e1cf8605b31037effed5de2a2fbbccba58d7b0a005ae7","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:28.428792+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/ccsb-scripps/AutoDock-Vina/blob/3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645/README.md","version":"3c65c0b3e6c2c1d183f6a175ecb65e3c5ba91645"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-ccsb-scripps-autodock-vina","kind":"source","links":[],"name":"ccsb-scripps/AutoDock-Vina official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"0f2074b432ba2516d0ea05fafab833e953201a3fbe0129c5b6beed9f5f089e81","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:21.083930+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/darlednik/GENEB/blob/9642d481e40c0af23995dcd162b779613f789f97/README.md","version":"9642d481e40c0af23995dcd162b779613f789f97"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-darlednik-geneb","kind":"source","links":[],"name":"darlednik/GENEB official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"772ebe52d2ba5100a28a888910c6f0c9fd4ded1d1372e3d89f6f1c48707e0365","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:25.524269+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/dauparas/ProteinMPNN/blob/8907e6671bfbfc92303b5f79c4b5e6ce47cdef57/README.md","version":"8907e6671bfbfc92303b5f79c4b5e6ce47cdef57"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-dauparas-proteinmpnn","kind":"source","links":[],"name":"dauparas/ProteinMPNN official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"2ea33af266b4268a55fd750d0f3265cd3165d61f6be375c5ea3b3ff5c58c7c8c","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:27.907349+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/DerrickWood/kraken2/blob/8c190b1b668825935dbf6dee5f969227dc8269bb/README.md","version":"8c190b1b668825935dbf6dee5f969227dc8269bb"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-derrickwood-kraken2","kind":"source","links":[],"name":"DerrickWood/kraken2 official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"6e404699412cb687bd737c4423c568f6ece1bad9f62a13e15ac1785062ccccd1","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:32.435757+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/drorlab/atom3d/blob/4c2f3b7e9efe128791b83f03b2e8cae91e78b018/README.md","version":"4c2f3b7e9efe128791b83f03b2e8cae91e78b018"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-drorlab-atom3d","kind":"source","links":[],"name":"drorlab/atom3d official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"74a897f0e97e3d4256f7cff11424dd264df4daaf8cabd9a6fb74b661421de038","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:25.632814+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/evolutionaryscale/esm/blob/bf343ba264b650dff7a073643725f9aaa1fdbe8d/README.md","version":"bf343ba264b650dff7a073643725f9aaa1fdbe8d"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-evolutionaryscale-esm","kind":"source","links":[],"name":"evolutionaryscale/esm official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"8b273c21a322fc9473d1b68d0dd40c8166ab2f89e4a190aa26ca87251b97cba9","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:24.844614+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/facebookresearch/esm/blob/2b369911bb5b4b0dda914521b9475cad1656b2ac/README.md","version":"2b369911bb5b4b0dda914521b9475cad1656b2ac"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-facebookresearch-esm","kind":"source","links":[],"name":"facebookresearch/esm official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"cd991ae6e76a5e84ea5449f91c4ed86ba4f57942682dc3c50d872c366bbbd4b7","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.381289+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://flip.protein.properties/","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-flip2","kind":"source","links":[],"name":"flip2 official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"a40358504726ee4e086f0623348fb2206c24ab8166d04a83d944579e9c62bdc8","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:20.985691+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/frederikkemarin/BEND/blob/ac6e80c75e09d83cf47a7b4bcf0e44599c5706cf/README.md","version":"ac6e80c75e09d83cf47a7b4bcf0e44599c5706cf"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-frederikkemarin-bend","kind":"source","links":[],"name":"frederikkemarin/BEND official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"5237cd1af3f1d7b9cf07cfe3cf722987bb27e10ee356ec108584943446bbb836","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:24.094166+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/GlycanML/GlycanML/blob/9f392aa6f9c6d74a296a250199beb347923d04e0/README.md","version":"9f392aa6f9c6d74a296a250199beb347923d04e0"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-glycanml-glycanml","kind":"source","links":[],"name":"GlycanML/GlycanML official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"cc16ba436ea8a967c764ac034a655d632b4c1ef8b4065868b8d07914c7cb88be","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:25.220155+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/google-deepmind/alphafold3/blob/c0f97eda2f1f482fd94d3a38bece18c7069b4a5c/README.md","version":"c0f97eda2f1f482fd94d3a38bece18c7069b4a5c"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-google-deepmind-alphafold3","kind":"source","links":[],"name":"google-deepmind/alphafold3 official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"8e5203afe343100832391e6155c7112f15cfe60bf0c21681d64e3420f854ef4d","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:24.355879+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/illumina/SpliceAI/blob/03f42437aaf56dc5dfd822c4ccee5aec1a705079/README.md","version":"03f42437aaf56dc5dfd822c4ccee5aec1a705079"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-illumina-spliceai","kind":"source","links":[],"name":"illumina/SpliceAI official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"9f51bbb20c4c5c36e77fb03ca1c5c36236e287c48a1ee31f53150545d421ec25","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:27.121929+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/instadeepai/nucleotide-transformer/blob/2dc37b86e16a6970fbc731751f7719d9f676f7f9/README.md","version":"2dc37b86e16a6970fbc731751f7719d9f676f7f9"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-instadeepai-nucleotide-transformer","kind":"source","links":[],"name":"instadeepai/nucleotide-transformer official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"f6e3b46a5f6744806846ccb4a054bcf3bec3da3d4249ada42fb9acac2a32024d","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:21.863141+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/J-SNACKKB/FLIP/blob/62cace8735f5610e2743cf06ce0f944b37fffaa6/README.md","version":"62cace8735f5610e2743cf06ce0f944b37fffaa6"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-j-snackkb-flip","kind":"source","links":[],"name":"J-SNACKKB/FLIP official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"79149435313841e6f9cc0c8b581259f3952a2469cbe6f9a733f7143b557025c2","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:25.416237+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/jwohlwend/boltz/blob/b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc/README.md","version":"b1ebfc46ecf57f5414e0d1a6f9027bbb122c53bc"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-jwohlwend-boltz","kind":"source","links":[],"name":"jwohlwend/boltz official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"058c4f98218ea2ed854681126b6f682c9f3beec91275781fb37e39c55d0c92ea","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:27.172955+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/kundajelab/chrombpnet/blob/09938fdb4397ec0006510e5251e48920a505d4de/README.md","version":"09938fdb4397ec0006510e5251e48920a505d4de"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-kundajelab-chrombpnet","kind":"source","links":[],"name":"kundajelab/chrombpnet official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"d21ee86b56c9b794f5b58f3c39c3e27c51d027a3b280d848457abf53f521f052","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:21.153480+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/kundajelab/DART-Eval/blob/af2a86d666c35304257c2fa7e15180e1fbcabb01/README.md","version":"af2a86d666c35304257c2fa7e15180e1fbcabb01"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-kundajelab-dart-eval","kind":"source","links":[],"name":"kundajelab/DART-Eval official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"c48be391c26a1ba35004265169c85661c4a96b0089d86383e922e25f6432a6f7","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:55.415067+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://fiehnlab.ucdavis.edu/projects/LipidBlast/","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-lipidblast","kind":"source","links":[],"name":"lipidblast official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"cfd3a044e568f9465d4365d8aa5e182041720301d70006e9f8f561608c660e70","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.464572+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://www.lipidmaps.org/resources/tools/lipidfinder/","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-lipidfinder","kind":"source","links":[],"name":"lipidfinder official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"8bcbe6e0632af4072d5124064e4032f814894185afcd684d85248836349f155c","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.217588+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://www.lipidmaps.org/databases/standardspectraDB/overview","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-lipidmaps-spectra","kind":"source","links":[],"name":"lipidmaps-spectra official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"71fee661aba53ea2285b776d572982a57c8637275687dfe4c5b7f5113a807d0e","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:23.520671+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/maabuu/posebusters/blob/6236d07017493531851cce775e8ef834d4763d2f/README.md","version":"6236d07017493531851cce775e8ef834d4763d2f"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-maabuu-posebusters","kind":"source","links":[],"name":"maabuu/posebusters official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"734a8cec5f667d74d421bf3b273ad7e256216109636da45aa7ceba21cd34de16","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:20.936532+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/MAGICS-LAB/DNABERT_2/blob/f25bed9ee20db966dff39e5c1571249d04e36404/README.md","version":"f25bed9ee20db966dff39e5c1571249d04e36404"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-magics-lab-dnabert-2","kind":"source","links":[],"name":"MAGICS-LAB/DNABERT_2 official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"3d002564045d3c493f30386df7983ee34b0f2f7ac89bf32ac63a5f780395d811","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:22.733154+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/mahmoodlab/HEST/blob/3ddb5eaf5bd2a8133e0c0e8015816489a3d99dc3/README.md","version":"3ddb5eaf5bd2a8133e0c0e8015816489a3d99dc3"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-mahmoodlab-hest","kind":"source","links":[],"name":"mahmoodlab/HEST official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"06a0c6b7adf444d7af53d20ce94c3cef483bd7be8db02fd8960c0704680d61ee","licence":null,"locator":"README.rst","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:29.085499+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/matchms/matchms/blob/066608589587c8d089afd2e8d55ceadb2766ea62/README.rst","version":"066608589587c8d089afd2e8d55ceadb2766ea62"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-matchms-matchms","kind":"source","links":[],"name":"matchms/matchms official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"fa39bffca31211baedb3b63df5dede775fcffff7b411d4261d5b18f526ae1153","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:27.482216+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/matsui-lab/GlycanGT/blob/96611518c971deb89215ca163deaf9de3a59fa32/README.md","version":"96611518c971deb89215ca163deaf9de3a59fa32"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-matsui-lab-glycangt","kind":"source","links":[],"name":"matsui-lab/GlycanGT official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"c0acb8f22d51b718d5c3a253e81b9882d804fa2aa153b6fcf3e826e22f085dd3","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:56.576460+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://meme-suite.org/meme/tools/fimo","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-meme","kind":"source","links":[],"name":"meme official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"e2dacff9ad56bca50c31373a1e87041eef948721b8aa0be001a95185c375c15f","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:23.885647+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/mims-harvard/TDC/blob/c310c35f27e3f506411018ac43d97b8ba23ca652/README.md","version":"c310c35f27e3f506411018ac43d97b8ba23ca652"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-mims-harvard-tdc","kind":"source","links":[],"name":"mims-harvard/TDC official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"926f0f196439564b095cbabe65f6a0acee3f22afbb0f6c8fe50bb3964082d15b","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:20.990833+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/ML-Bioinfo-CEITEC/genomic_benchmarks/blob/605d8539830e16c85abe7826990958303ffc5e1c/README.md","version":"605d8539830e16c85abe7826990958303ffc5e1c"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-ml-bioinfo-ceitec-genomic-benchmarks","kind":"source","links":[],"name":"ML-Bioinfo-CEITEC/genomic_benchmarks official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"f9f1c1d62adc471661ca98b30c0250e9f3ce0cff7433830f149f5f48ea41c3da","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:26.372370+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/ml4bio/RNA-FM/blob/348951516e0963d22bbb33b3c9fc18c89081d38e/README.md","version":"348951516e0963d22bbb33b3c9fc18c89081d38e"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-ml4bio-rna-fm","kind":"source","links":[],"name":"ml4bio/RNA-FM official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"f0c67304e20ced42938829dfee39480cef51ee3a4357ee8b53eafc89c305fa60","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:21.823561+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/morrislab/mRNABench/blob/74f96b8e6ae9f41cc3cccff089d826a62d5604b8/README.md","version":"74f96b8e6ae9f41cc3cccff089d826a62d5604b8"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-morrislab-mrnabench","kind":"source","links":[],"name":"morrislab/mRNABench official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"e9c0b76d743af53198b0197bfa58305bf26cff822538658d0366880fcf56a8a9","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:21.898594+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/mrzzmrzz/NABench/blob/99c8681ec1eab706e10ff90a5c329dcf184cc1d1/README.md","version":"99c8681ec1eab706e10ff90a5c329dcf184cc1d1"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-mrzzmrzz-nabench","kind":"source","links":[],"name":"mrzzmrzz/NABench official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"e503f496841fd1d3836ebdab573de645257fc92fe7b5479056592da512e0ef3f","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.933176+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://arxiv.org/abs/2605.19752","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-msalign","kind":"source","links":[],"name":"msalign official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"b9e620179f6b9a9aafd9eaf8b874b8f1fa2c8c4ffdced39f2e075391ec3d476d","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:24.835167+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/Murali-group/Beeline/blob/37464085eb8a95d6cc6a3d3a3c649d36db6052ed/README.md","version":"37464085eb8a95d6cc6a3d3a3c649d36db6052ed"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-murali-group-beeline","kind":"source","links":[],"name":"Murali-group/Beeline official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"321487a8de52c6cfa647a658f61150dd72e0acb0125470524fb94d1f8b23321a","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:21.811694+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/OATML-Markslab/ProteinGym/blob/144fe22b07dfaeec2b366f2346203a9838a55b4c/README.md","version":"144fe22b07dfaeec2b366f2346203a9838a55b4c"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-oatml-markslab-proteingym","kind":"source","links":[],"name":"OATML-Markslab/ProteinGym official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"ae2da4ad3ae1d822e0010b6c023b26ef1d02b97007e17611dd4b1e22c70d4998","licence":null,"locator":"README.rst","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:28.394293+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/opencobra/cobrapy/blob/5aa19300fbf5dd632a9ac5c39ca28c1c621b943d/README.rst","version":"5aa19300fbf5dd632a9ac5c39ca28c1c621b943d"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-opencobra-cobrapy","kind":"source","links":[],"name":"opencobra/cobrapy official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"ee26d791b60868c3e7701357831a49d8325f8d4d7cb0769b5177e5a67713114b","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:22.429615+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/openproblems-bio/openproblems/blob/0ca5d0cd040b741c1b6cc2e4cad7230cb2c50131/README.md","version":"0ca5d0cd040b741c1b6cc2e4cad7230cb2c50131"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-openproblems-bio-openproblems","kind":"source","links":[],"name":"openproblems-bio/openproblems official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"1d53c3b89030dc4651d3e7bf4749256b7e579330fc7a660cffaa992d646da34a","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:23.659162+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/plinder-org/plinder/blob/85b3f1cb1763530a6cfd934f4263a1777c41afa4/README.md","version":"85b3f1cb1763530a6cfd934f4263a1777c41afa4"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-plinder-org-plinder","kind":"source","links":[],"name":"plinder-org/plinder official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"a6afe934f9894cc71fb9b561c7a122d043b377c27b3fc6f18505fe65b6226256","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:27.232940+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/pluskal-lab/DreaMS/blob/dbec3a0b514a99e5056cfccde4559fda8cfe8129/README.md","version":"dbec3a0b514a99e5056cfccde4559fda8cfe8129"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-pluskal-lab-dreams","kind":"source","links":[],"name":"pluskal-lab/DreaMS official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"08bf3607e6e2e5462b81eac85d0e71d9d23ce1c9bf1a370c9d1079ecd60ee2d8","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:23.954980+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/pluskal-lab/MassSpecGym/blob/f259fe3780d5bd227fc6ece36ce6f397c2eef716/README.md","version":"f259fe3780d5bd227fc6ece36ce6f397c2eef716"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-pluskal-lab-massspecgym","kind":"source","links":[],"name":"pluskal-lab/MassSpecGym official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"2e488850a6557bb57407615f2df9194351718b3dc0298a03c0c97d8e93460712","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.349409+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://proteinbench.github.io/","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-proteinbench","kind":"source","links":[],"name":"proteinbench official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"d89eb9790ce805ce23e1b1a6804f4d67c81056025e4852b875f8ddfd87a08125","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:25.975464+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/RosettaCommons/RFdiffusion/blob/86507b6538f51fce57b5a72477165f03999ed7ae/README.md","version":"86507b6538f51fce57b5a72477165f03999ed7ae"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-rosettacommons-rfdiffusion","kind":"source","links":[],"name":"RosettaCommons/RFdiffusion official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"eb46b8a54e60643ca0cd8cb375ede05b01dcbd2380ca17ce8027a92ba13cebbb","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:26.353619+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/scverse/scvi-tools/blob/73b28e44223621470e582a81a102c107bb22678b/README.md","version":"73b28e44223621470e582a81a102c107bb22678b"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-scverse-scvi-tools","kind":"source","links":[],"name":"scverse/scvi-tools official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"fe78b7b1ede50c673b8f0d91fb2637707685c78b269ee157edb6cbc4d510d61b","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:26.451955+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/snap-stanford/GEARS/blob/f374e43e197b295016d80395d7a54ddb81cc6769/README.md","version":"f374e43e197b295016d80395d7a54ddb81cc6769"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-snap-stanford-gears","kind":"source","links":[],"name":"snap-stanford/GEARS official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"2f8690d6a4a9767973ba9ea2af019d43d6108555ef0e9a69786033a71547d133","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:28.898945+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/soedinglab/hh-suite/blob/43095e46ada4ec2a8a47d47ef5ad7e38b1429f7b/README.md","version":"43095e46ada4ec2a8a47d47ef5ad7e38b1429f7b"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-soedinglab-hh-suite","kind":"source","links":[],"name":"soedinglab/hh-suite official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"b6c591a763bf99c027857385f0e87ce5aa96caeaa74d71afd1fcec449eadb3d7","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:28.691617+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/soedinglab/MMseqs2/blob/d401e78c2d18a822cdb1527d7464a043f6035a15/README.md","version":"d401e78c2d18a822cdb1527d7464a043f6035a15"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-soedinglab-mmseqs2","kind":"source","links":[],"name":"soedinglab/MMseqs2 official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"a384ec62c86d568aeb0fe3e8a3b111f071fbcfacb937b407eb6da24fc70c823f","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:25.578919+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/sokrypton/ColabFold/blob/84c27d9cc500489fd9b97545d2325b9d00f251d5/README.md","version":"84c27d9cc500489fd9b97545d2325b9d00f251d5"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-sokrypton-colabfold","kind":"source","links":[],"name":"sokrypton/ColabFold official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"b28c74fe3cd6b69a8ba6d84891d0539e54dfef882abd5ed4d11ed0b029bb477a","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:32.443187+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/songlab-cal/tape/blob/6d345c2b2bbf52cd32cf179325c222afd92aec7e/README.md","version":"6d345c2b2bbf52cd32cf179325c222afd92aec7e"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-songlab-cal-tape","kind":"source","links":[],"name":"songlab-cal/tape official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"0403f84453aace301c7d02895d94a977ccbbec77c3a94a49a23d0d529dd48d31","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:21.025462+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/terry-r123/RNABenchmark/blob/da7f9c7ac3f39605af27e1dfcdf879adba963d79/README.md","version":"da7f9c7ac3f39605af27e1dfcdf879adba963d79"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-terry-r123-rnabenchmark","kind":"source","links":[],"name":"terry-r123/RNABenchmark official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"db7aa3a701778d541bf47d8214b50e1ce9f73e92dc3e810bfd740270e9e353aa","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:32.426513+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/theislab/scib/blob/cd67913396b4c0430710b3d90f1d1841f5fa4468/README.md","version":"cd67913396b4c0430710b3d90f1d1841f5fa4468"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-theislab-scib","kind":"source","links":[],"name":"theislab/scib official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"9117f9d255d6b6e810d224a600d381192bccccd3f0cd417fed9559d49ca8fffd","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:24.697521+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/tkzeng/Pangolin/blob/5cf94b8db938c658391b4305cd7ce33297d44ff7/README.md","version":"5cf94b8db938c658391b4305cd7ce33297d44ff7"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-tkzeng-pangolin","kind":"source","links":[],"name":"tkzeng/Pangolin official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"57dd9cbd5e6b62ed665eebd909d82ffec580a55c3bd2f48a928f807a8c983b1e","licence":null,"locator":"HTML main page","missing_metadata":{"immutable_source_version":"unavailable","licence":"unextracted","version":"unreported"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:31:54.531595+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","scope_note":"Mutable official page: content hash and retrieval time recorded, but the original page is not redistributed.","url":"https://arcinstitute.org/news/virtual-cell-challenge-2026","version":null},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-vcc2026","kind":"source","links":[],"name":"vcc2026 official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"d37146b01e4273062a5230c496a9af8414ee7ef14fcd905cf415959256d59c4e","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:26.326355+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/ViennaRNA/ViennaRNA/blob/1ffec79f5e258896160f7362ced8263450f371dc/README.md","version":"1ffec79f5e258896160f7362ced8263450f371dc"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-viennarna-viennarna","kind":"source","links":[],"name":"ViennaRNA/ViennaRNA official source","source_ids":[],"status":"source_checked"} {"attributes":{"artifact_sha256":"d8522e585806ec0008a36558e0dd1f6f0b0bb4deffb3f64a7ee4f09cd087e9f4","licence":null,"locator":"README.md","missing_metadata":{"licence":"unextracted"},"publication_status":"official_project_source","retrieved_at":"2026-09-16T10:30:23.079976+00:00","review_scope":"URL and retrieved artifact identity checked; individual scientific claims need their own review.","url":"https://github.com/Virtual-Cell-Research-Community/scPertEval/blob/4685f11927e887745737600170da7a655b727553/README.md","version":"4685f11927e887745737600170da7a655b727553"},"description":"Primary project documentation or project-maintained evidence.","facets":{},"id":"src-discovery-virtual-cell-research-community-scperteval","kind":"source","links":[],"name":"Virtual-Cell-Research-Community/scPertEval official source","source_ids":[],"status":"source_checked"} {"id":"structure-informed-plm-2025","kind":"source","name":"Structure-Informed Protein Language Models are Robust Predictors for Variant Effects","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/","version":"Human Genetics 2025 journal article (online 2024)","retrieved_at":"2026-09-15T23:33:26Z","doi":"10.1007/s00439-024-02695-w","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"76082e1cd992d2c09c38f86d05aba575cc76c5022b53a297123b713bb1ce9267","artifact_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/","artifact_retrieved_at":"2026-09-16T10:45:41.099916+00:00","legacy_paper":{"id":"structure-informed-plm-2025","title":"Structure-Informed Protein Language Models are Robust Predictors for Variant Effects","year":2025,"publication_status":"peer_reviewed","version":"Human Genetics 2025 journal article (online 2024)","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12068927/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:33:26Z","doi":"10.1007/s00439-024-02695-w","notes":"Final Human Genetics Table 4, AA+SS+RSA+CM AUROC .803 checked directly; Research Square preprint Table 3 prints the same value."},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"transbind-2026","kind":"source","name":"Integrating protein and DNA embeddings for improving genome-wide transcription factor binding site prediction","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["molecular-interactions"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13145115/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1093/nargab/lqag047","publication_status":"peer_reviewed","year":2026,"artifact_sha256":"5d777f5925e941b7d087035d5d87e79ef75ae8d6456a770ffe8c527da566fee0","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC13145115/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:58.585Z","legacy_paper":{"id":"transbind-2026","title":"Integrating protein and DNA embeddings for improving genome-wide transcription factor binding site prediction","year":2026,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC13145115/","primary_domain":"molecular-interactions","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: NAR Genomics and Bioinformatics; PMC ID: PMC13145115. Protein-DNA model; comparator scores in table not copied into this batch.","doi":"10.1093/nargab/lqag047"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"tu-fold-2025","kind":"source","name":"RNA secondary structure prediction by conducting multi-class classifications","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["rna-transcriptomes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12008525/","version":"version of record","retrieved_at":"2026-09-15T23:25:00Z","doi":"10.1016/j.csbj.2025.04.001","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"5aa376d6466daee83fc307baa39fd48da0f185ff30a178624025032d4cbe597d","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12008525/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:16.505799+00:00","legacy_paper":{"id":"tu-fold-2025","title":"RNA secondary structure prediction by conducting multi-class classifications","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12008525/","primary_domain":"rna-transcriptomes","retrieved_utc":"2026-09-15T23:25:00Z","notes":"Primary full text via Europe PMC XML; venue: Computational and Structural Biotechnology Journal; PMC ID: PMC12008525.","doi":"10.1016/j.csbj.2025.04.001"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"vaxign-esm-2024","kind":"source","name":"Enhancing Vaxign-DL for Vaccine Candidate Prediction with added ESM-Generated Features","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11398487/","version":"preprint version in PMC","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.1101/2024.09.04.611295","publication_status":"preprint","year":2024,"artifact_sha256":"b76fff917addd0e9ff8a2fc843496132ecf832d3ceef248e3edf2dbc78baab5f","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC11398487/fullTextXML","artifact_retrieved_at":"2026-09-16T10:41:06Z","legacy_paper":{"id":"vaxign-esm-2024","title":"Enhancing Vaxign-DL for Vaccine Candidate Prediction with added ESM-Generated Features","year":2024,"publication_status":"preprint","version":"preprint version in PMC","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC11398487/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: bioRxiv; PMC ID: PMC11398487. Preprint; combined classifier uses ESM features rather than ESM-only predictions.","doi":"10.1101/2024.09.04.611295"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"viral-contig-simulation-2021","kind":"source","name":"Simulation study and comparative evaluation of viral contiguous sequence identification tools","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["microbes-communities"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8207588/","version":"version of record","retrieved_at":"2026-09-15T23:29:32Z","doi":"10.1186/s12859-021-04242-0","publication_status":"peer_reviewed","year":2021,"artifact_sha256":"93a24652edfa6d9f686862479df3addf50a2d5d432d2d5a31882075fa59dfdfd","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC8207588/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:50.134Z","legacy_paper":{"id":"viral-contig-simulation-2021","title":"Simulation study and comparative evaluation of viral contiguous sequence identification tools","year":2021,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC8207588/","primary_domain":"microbes-communities","retrieved_utc":"2026-09-15T23:29:32Z","notes":"Primary full text verified using Europe PMC XML; venue: BMC Bioinformatics; PMC ID: PMC8207588.","doi":"10.1186/s12859-021-04242-0"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}} {"id":"viral-immune-mimicry-2025","kind":"source","name":"Protein Language Models Expose Viral Immune Mimicry","description":"Primary paper retained with its original identifier. Metadata inherited from the literature collection; individual result checks are separate.","status":"discovered","facets":{"areas":["proteins-complexes"]},"source_ids":[],"links":[],"attributes":{"url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12474240/","version":"version of record","retrieved_at":"2026-09-15T23:37:05Z","doi":"10.3390/v17091199","publication_status":"peer_reviewed","year":2025,"artifact_sha256":"15250af2f75f70e2b6a3725d00bc7e276ed9d2d4a54f0ae9d6eaabf6be13e4a1","artifact_url":"https://www.ebi.ac.uk/europepmc/webservices/rest/PMC12474240/fullTextXML","artifact_retrieved_at":"2026-09-16T10:33:57.274Z","legacy_paper":{"id":"viral-immune-mimicry-2025","title":"Protein Language Models Expose Viral Immune Mimicry","year":2025,"publication_status":"peer_reviewed","version":"version of record","source_url":"https://pmc.ncbi.nlm.nih.gov/articles/PMC12474240/","primary_domain":"proteins-complexes","retrieved_utc":"2026-09-15T23:37:05Z","notes":"Primary full text verified via Europe PMC fullTextXML; venue: Viruses; PMC ID: PMC12474240. Downstream classifier uses ESM2 representations; table does not report a pure zero-shot language-model score.","doi":"10.3390/v17091199"},"scope_decision":"included","missing_metadata":{"licence":"not_reported_in_legacy_extract"}}}