{"id":"W2589068952","doi":"10.29173/cais150","title":"Performance in ART1-Like Document Clustering with Variable Similarity Thresholds","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Data Management and Algorithms","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Cluster analysis; Similarity (geometry); Variable (mathematics); Implementation; Massively parallel; Document clustering; Data mining; Information retrieval; Parallel computing; Artificial intelligence; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":["scholarly_communication"],"category_scores_codex":[0.0006106369,0.0002468345,0.000350273,0.0001723627,0.0001048681,0.003439159,0.003000047,0.00007858735,0.00002729227],"category_scores_gemma":[0.0006252163,0.0001770318,0.00005537223,0.0006636067,0.0001818296,0.02442624,0.001679034,0.0002483501,0.000006036883],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005155356,"about_ca_system_score_gemma":0.000113767,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003132421,"about_ca_topic_score_gemma":0.00002252618,"domain_scores_codex":[0.9982008,0.00001107856,0.0004091638,0.000405122,0.0004973122,0.0004765176],"domain_scores_gemma":[0.9906831,0.0000476609,0.0003997921,0.0003522474,0.008424537,0.00009264171],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0001507139,0.0005144146,0.8635796,0.00140734,0.0002018269,0.000005402519,0.02568161,0.0004678767,0.005954002,0.05807335,0.005510662,0.03845325],"study_design_scores_gemma":[0.002942738,0.001385162,0.7181647,0.001743786,0.00009903953,0.00007025238,0.002026556,0.1901758,0.01791123,0.01448355,0.04952789,0.001469258],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9811269,0.00003035309,0.002056996,0.001414766,0.0001495248,0.0005389596,0.00001437155,0.00006182109,0.01460628],"genre_scores_gemma":[0.9885807,0.00005576217,0.01046653,0.0002271632,0.00003271024,0.00005750315,0.000002135218,0.00001231177,0.000565181],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1897079,"threshold_uncertainty_score":0.9975954,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01507865651523827,"score_gpt":0.2154069153870535,"score_spread":0.2003282588718152,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}