{"id":"W4410878274","doi":"10.1016/j.buildenv.2025.113244","title":"Consistency is key: Evaluation of and recommendations for thermostat usability testing","year":2025,"lang":"en","type":"article","venue":"Building and Environment","topic":"Gaze Tracking and Assistive Technology","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Carleton University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Usability; Key (lock); Consistency (knowledge bases); Thermostat; Computer science; Engineering; Reliability engineering; Human–computer interaction; Computer security; Mechanical engineering; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2095065,0.001094918,0.001624222,0.00308098,0.002403204,0.007362089,0.003679074,0.002936126,0.004747294],"category_scores_gemma":[0.5629951,0.0009886106,0.001982637,0.00173695,0.002252131,0.008719933,0.003078013,0.002997319,0.002053356],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003999368,"about_ca_system_score_gemma":0.01048433,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004452546,"about_ca_topic_score_gemma":0.01020038,"domain_scores_codex":[0.7290949,0.1558939,0.03274025,0.006741667,0.07295948,0.002569907],"domain_scores_gemma":[0.3328286,0.4227365,0.03103419,0.03641148,0.1716924,0.005296817],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.002944999,0.002923577,0.1334587,0.005232071,0.0006727498,0.0002271515,0.01394269,0.001954095,0.01138822,0.009082405,0.03919621,0.7789771],"study_design_scores_gemma":[0.0027446,0.02835418,0.5783876,0.03635216,0.003501198,0.001697669,0.04009408,0.05057403,0.06196506,0.03951065,0.1549509,0.001867867],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4747924,0.01384169,0.3314911,0.05485428,0.004958044,0.02010203,0.001664479,0.006711542,0.09158441],"genre_scores_gemma":[0.6974714,0.001650767,0.2823125,0.004083232,0.0003340681,0.006177239,0.0007301515,0.0007741334,0.006466416],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.2095065,"threshold_uncertainty_score":0.9748193,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06742124919904682,"score_gpt":0.3164235165436944,"score_spread":0.2490022673446476,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}