{"id":"W3197583449","doi":"10.1016/j.aca.2021.339001","title":"Advanced data fusion: Random forest proximities and pseudo-sample principle towards increased prediction accuracy and variable interpretation","year":2021,"lang":"en","type":"article","venue":"Analytica Chimica Acta","topic":"Spectroscopy and Chemometric Analyses","field":"Chemistry","cited_by":17,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek","keywords":"Random forest; Principal component analysis; Sensor fusion; Data mining; Data set; Fusion; Sample (material); Pattern recognition (psychology); Artificial intelligence; Computer science; Chemistry","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001954866,0.0002341014,0.0004131435,0.0001086841,0.0002056908,0.0001867319,0.0002865924,0.0001351784,0.0008980411],"category_scores_gemma":[0.003167003,0.0002183078,0.0000512523,0.0004605092,0.0001168946,0.000595262,0.0004950861,0.0002335765,0.000001883469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00005745744,"about_ca_system_score_gemma":0.0001855072,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00009074102,"about_ca_topic_score_gemma":0.00005118096,"domain_scores_codex":[0.9983745,0.00003540384,0.0003838281,0.000663614,0.0002572532,0.0002854649],"domain_scores_gemma":[0.9980333,0.0006981703,0.0001667941,0.0008069871,0.0001344819,0.0001602086],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002570124,0.001077453,0.07594872,0.001784049,0.002520569,0.00004378212,0.001228827,0.00002502697,0.8996129,0.003484936,0.004081206,0.007622406],"study_design_scores_gemma":[0.01311904,0.0002409796,0.02102932,0.0005570141,0.005104622,0.0003352775,0.003858993,0.557975,0.3590536,0.009568001,0.02744753,0.001710637],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9894649,0.0002810463,0.001509376,0.000671759,0.00004634983,0.000148667,0.0006532666,0.0001387523,0.007085828],"genre_scores_gemma":[0.9905856,0.001069547,0.005791368,0.0001573278,0.0001207158,0.0000213856,0.001675197,0.00002483914,0.0005540245],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.55795,"threshold_uncertainty_score":0.983292,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01880636204416096,"score_gpt":0.2853355387915155,"score_spread":0.2665291767473545,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}