{"id":"W4403434240","doi":"10.26434/chemrxiv-2024-rrbhc","title":"Composition and structure analyzer/featurizer for explainable machine-learning models to predict solid state structures","year":2024,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Python (programming language); Computer science; Spectrum analyzer; Artificial intelligence; Computational science; Machine learning; Algorithm; Open source; Data mining; Software; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007262138,0.00123327,0.0004975524,0.0007999689,0.0005642854,0.0005711504,0.00161467,0.0007690781,0.01187779],"category_scores_gemma":[0.001814973,0.0005003838,0.0009553322,0.0006127274,0.0004024104,0.0009946896,0.0007796443,0.001858773,0.002630752],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005476346,"about_ca_system_score_gemma":0.001242236,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002147716,"about_ca_topic_score_gemma":0.004007847,"domain_scores_codex":[0.9998385,0.00004474949,0.00000851528,0.00003574228,0.00005845296,0.00001419305],"domain_scores_gemma":[0.9996928,0.0001768315,0.00002462296,0.00005002233,0.00004368641,0.00001192149],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003327043,0.0003079621,0.004070282,0.0005801001,0.0002232027,0.0004034013,0.0002008125,0.6018042,0.03041985,0.08959328,0.03381749,0.2382467],"study_design_scores_gemma":[0.00001127009,0.00001127189,0.0001370736,0.000007469734,0.000007271863,0.00002425651,0.000007427064,0.9776394,0.006343954,0.01082583,0.004978368,0.000006304084],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01130743,0.00007184539,0.9596875,0.0001620375,0.00004467738,0.0001023667,0.001274199,0.02492665,0.002423383],"genre_scores_gemma":[0.1104262,0.0001518611,0.8790606,0.0001280512,0.00003530517,0.0005982906,0.002992961,0.003490028,0.003116684],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01187779,"threshold_uncertainty_score":0.03973514,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0183463639295776,"score_gpt":0.2740470386822784,"score_spread":0.2557006747527008,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}