{"id":"W4382937982","doi":"10.21203/rs.3.rs-3083229/v1","title":"Anchor Clustering for million-scale immune repertoire sequencing data","year":2023,"lang":"en","type":"preprint","venue":"Research Square","topic":"vaccines and immunoinformatics approaches","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"University of Guelph","keywords":"Cluster analysis; Repertoire; Pairwise comparison; Single-linkage clustering; Computer science; Correlation clustering; CURE data clustering algorithm; Consensus clustering; Data mining; Computational biology; Artificial intelligence; Biology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science"],"consensus_categories":[],"category_scores_codex":[0.00244641,0.0003028361,0.0003663903,0.0002009288,0.0003905942,0.0002462976,0.001728705,0.0005271636,0.00001152191],"category_scores_gemma":[0.0007031983,0.0002831955,0.0002087057,0.0001972711,0.00008317578,0.00001302537,0.008394822,0.0007122474,0.00002892556],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001118175,"about_ca_system_score_gemma":0.0005256568,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002651902,"about_ca_topic_score_gemma":0.0001814745,"domain_scores_codex":[0.9971633,0.0001219063,0.0005669915,0.0009166631,0.0004601755,0.0007709911],"domain_scores_gemma":[0.9959728,0.00008058656,0.0001505134,0.003198538,0.0004758996,0.0001216697],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001463105,0.0003950669,0.002165106,0.02918608,0.001770911,0.00004402481,0.002860699,0.01386834,0.6710276,0.000137699,0.2114375,0.06564388],"study_design_scores_gemma":[0.003930224,0.002127111,0.004794525,0.005745509,0.0001714298,0.00009212775,0.01409841,0.4770989,0.1417802,0.002009993,0.3447613,0.003390277],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8550817,0.03679116,0.07280904,0.004984823,0.003562117,0.01312646,0.009227661,0.000454758,0.003962243],"genre_scores_gemma":[0.8979328,0.01177731,0.02332082,0.00009922928,0.00315266,0.001737057,0.05006459,0.0004407872,0.01147476],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.5292474,"threshold_uncertainty_score":0.999962,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1866787123305273,"score_gpt":0.4017514002363556,"score_spread":0.2150726879058283,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}