{"id":"W2929353344","doi":"10.1109/access.2019.2908032","title":"Enhancing Predictive Power of Cluster-Boosted Regression With Text-Based Indexing","year":2019,"lang":"en","type":"article","venue":"IEEE Access","topic":"Machine Learning in Healthcare","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"King Mongkut's University of Technology Thonburi","keywords":"Computer science; Cluster analysis; Artificial intelligence; Search engine indexing; Bigram; Curse of dimensionality; Support vector machine; Word (group theory); Regression; Principal component analysis; Data mining; Machine learning; Regression analysis; Pattern recognition (psychology); Statistics; Mathematics; Trigram","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00217778,0.001486047,0.001900666,0.00218671,0.0005402399,0.00123852,0.00173732,0.0008946541,0.00191054],"category_scores_gemma":[0.007907973,0.0003550146,0.001302477,0.002597236,0.0003468271,0.001559722,0.001010269,0.001673051,0.001986825],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006877443,"about_ca_system_score_gemma":0.001262605,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01144481,"about_ca_topic_score_gemma":0.008626005,"domain_scores_codex":[0.9987289,0.0002821062,0.00008171514,0.0004600614,0.0003035142,0.0001437827],"domain_scores_gemma":[0.9972457,0.001271527,0.0002135244,0.0003751879,0.0008107954,0.00008334428],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008730796,0.0008183271,0.01146298,0.0003025515,0.0003133385,0.0002240253,0.0001338564,0.3105236,0.01539062,0.003254283,0.01696642,0.639737],"study_design_scores_gemma":[0.0000180629,0.00004674146,0.0008733425,0.000006608708,0.00002909804,0.00001940287,0.00001197614,0.9938087,0.002979831,0.001522593,0.0006716468,0.00001202187],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1171805,0.002148686,0.8547062,0.0009545384,0.0004479954,0.0002519587,0.002442225,0.01774001,0.004127976],"genre_scores_gemma":[0.7289266,0.0006770206,0.2583245,0.0003744655,0.0004414794,0.0002300709,0.006493026,0.0004180377,0.004114799],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01144481,"threshold_uncertainty_score":0.0227564,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01354817976175746,"score_gpt":0.3062812843852586,"score_spread":0.2927331046235011,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}