{"id":"W4391543065","doi":"10.1016/j.buildenv.2024.111268","title":"A comparative analysis of machine learning and statistical methods for evaluating building performance: A systematic review and future benchmarking framework","year":2024,"lang":"en","type":"review","venue":"Building and Environment","topic":"Building Energy and Comfort Optimization","field":"Engineering","cited_by":48,"is_retracted":false,"has_abstract":false,"ca_institutions":"Carleton University","funders":"","keywords":"Benchmarking; Computer science; Machine learning; Artificial intelligence; Statistical analysis; Engineering; Management science; Statistics; Mathematics; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.030302,0.002022912,0.0110872,0.01737628,0.0007273365,0.005286865,0.003331079,0.002962257,0.003285874],"category_scores_gemma":[0.07708593,0.001002899,0.01085482,0.01371561,0.001494838,0.003974284,0.002382128,0.002086824,0.0003288168],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004893698,"about_ca_system_score_gemma":0.01568875,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01056418,"about_ca_topic_score_gemma":0.02851891,"domain_scores_codex":[0.9820176,0.006198121,0.006436304,0.001306671,0.003750598,0.0002908922],"domain_scores_gemma":[0.92541,0.06133426,0.006637756,0.0009770562,0.005264456,0.0003763462],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0003760681,0.00005478215,0.001307189,0.8392249,0.01896919,0.00008082885,0.0003061515,0.000533523,0.0003259065,0.001644515,0.00176421,0.1354128],"study_design_scores_gemma":[0.0003324461,0.0005073123,0.004515362,0.8316368,0.127413,0.0003340716,0.0006415285,0.0006669795,0.0003901687,0.001823674,0.03163756,0.000100878],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.0005017021,0.9981247,0.0006006659,0.0002115082,0.00007250231,0.0001804335,0.0001587148,0.00000644677,0.0001434124],"genre_scores_gemma":[0.01106163,0.9843014,0.003332379,0.0004850341,0.00008256207,0.0004284792,0.0002086796,0.00001002315,0.00008976945],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.030302,"threshold_uncertainty_score":0.1602542,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03210962319066626,"score_gpt":0.3772872998526078,"score_spread":0.3451776766619415,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}