{"id":"W4407020881","doi":"10.2139/ssrn.5119484","title":"Consistency is Key: Evaluation of and Recommendations for Thermostat Usability Testing","year":2025,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Usability and User Interface Design","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Carleton University","funders":"","keywords":"Usability; Consistency (knowledge bases); Key (lock); Computer science; Thermostat; Human–computer interaction; Engineering; Computer security; Mechanical engineering; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.1954877,0.001281033,0.001543277,0.003180701,0.002261487,0.007838073,0.004373835,0.003625272,0.006039963],"category_scores_gemma":[0.5730489,0.001136749,0.001991106,0.002115068,0.002623884,0.00970124,0.003050572,0.003258529,0.002316427],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00389758,"about_ca_system_score_gemma":0.007949294,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00363156,"about_ca_topic_score_gemma":0.006393926,"domain_scores_codex":[0.7623975,0.1539687,0.0209812,0.005238392,0.05559321,0.001820961],"domain_scores_gemma":[0.3510801,0.4746405,0.0206235,0.03342243,0.1156723,0.004561226],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.005914874,0.002806745,0.0817527,0.005507202,0.0005744029,0.0002024104,0.01291924,0.002776546,0.01012141,0.01881745,0.04334281,0.8152643],"study_design_scores_gemma":[0.006135288,0.04222328,0.3926781,0.03742202,0.005113949,0.001741589,0.03987401,0.1221181,0.08741382,0.105348,0.1574804,0.002451411],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4295034,0.0142153,0.407129,0.04803283,0.004131194,0.01889436,0.001560793,0.00696344,0.06956974],"genre_scores_gemma":[0.6553976,0.00149129,0.3262926,0.003437487,0.000242703,0.005933263,0.0006333662,0.0008719719,0.005699687],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.1954877,"threshold_uncertainty_score":0.9921069,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1069850722421773,"score_gpt":0.3578284241006478,"score_spread":0.2508433518584705,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}