{"id":"W4386384752","doi":"10.48550/arxiv.2308.16744","title":"MS-BioGraphs: Sequence Similarity Graph Datasets","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Genomics and Phylogenetic Studies","field":"Biochemistry, Genetics and Molecular Biology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Engineering and Physical Sciences Research Council; Queen's University; Queen's University Belfast; Department for the Economy; UK Research and Innovation","keywords":"Computer science; Graph; Data mining; Theoretical computer science; Similarity (geometry); Artificial intelligence","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008430532,0.00139744,0.0006343286,0.004043002,0.001055152,0.0008546762,0.002421621,0.002182732,0.005325412],"category_scores_gemma":[0.005428493,0.0004445922,0.001090995,0.005902079,0.000545734,0.001488489,0.001423857,0.001850404,0.004935619],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001403029,"about_ca_system_score_gemma":0.001394404,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009161559,"about_ca_topic_score_gemma":0.01811978,"domain_scores_codex":[0.9990228,0.0001827167,0.0001087783,0.0003254117,0.0002727551,0.00008753734],"domain_scores_gemma":[0.9979966,0.0006183569,0.000257841,0.0005430268,0.0003378062,0.0002464439],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001167513,0.0007889016,0.01666655,0.003454579,0.0003890623,0.0009003609,0.000441533,0.02145996,0.01816491,0.01448168,0.8716376,0.05044739],"study_design_scores_gemma":[0.001194867,0.000500883,0.05323258,0.0003392242,0.0002687378,0.002478065,0.0008259744,0.0706299,0.02352029,0.04277732,0.804023,0.0002091256],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.07566833,0.002078325,0.01077612,0.001840204,0.0002115371,0.0003738055,0.8873387,0.01541466,0.006298314],"genre_scores_gemma":[0.03252416,0.0005625073,0.02147386,0.0002802107,0.00003596027,0.0003117501,0.9432295,0.0005372755,0.001044639],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.009161559,"threshold_uncertainty_score":0.01821649,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1076922189063325,"score_gpt":0.217498001834348,"score_spread":0.1098057829280155,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}