{"id":"W2809365483","doi":"10.1186/s12961-018-0334-9","title":"Validity and usability testing of a health systems guidance appraisal tool, the AGREE-HS","year":2018,"lang":"en","type":"article","venue":"Health Research Policy and Systems","topic":"Health Policy Implementation Science","field":"Health Professions","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University; Juravinski Cancer Centre","funders":"Canadian Institutes of Health Research","keywords":"Usability; Quality (philosophy); Context (archaeology); Reliability (semiconductor); Face validity; Computer science; Applied psychology; Test (biology); Health informatics; Content validity; Consistency (knowledge bases); Health services research; Psychology; Knowledge management; Medicine; Psychometrics; Public health; Human–computer interaction; Nursing; Clinical psychology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","sts"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1128247,0.0002400095,0.0008768784,0.000435033,0.006306539,0.0001125178,0.0005182636,0.0001777073,0.00001580201],"category_scores_gemma":[0.05246786,0.0001678305,0.00003348845,0.002088075,0.001775334,0.0002825593,0.000479776,0.001131798,0.00006180173],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009099047,"about_ca_system_score_gemma":0.01197212,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.1745042,"about_ca_topic_score_gemma":0.002171653,"domain_scores_codex":[0.9608909,0.03046558,0.00314224,0.0007537821,0.001875118,0.002872345],"domain_scores_gemma":[0.9520919,0.04084864,0.00176282,0.001336274,0.002589932,0.001370465],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001605184,0.0001452519,0.529582,0.04309978,0.00003261359,0.000002361679,0.1227787,0.00001336701,0.0002282413,0.1141955,0.1684811,0.02128064],"study_design_scores_gemma":[0.001832676,0.002755997,0.4323792,0.003415556,0.000004115028,0.0000661817,0.03888528,0.00976014,0.000007205851,0.0009967014,0.5095193,0.0003776111],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8731263,0.002361996,0.001261081,0.1069625,0.001687599,0.009312018,0.0004982877,0.0001435859,0.004646642],"genre_scores_gemma":[0.9906704,0.0003229831,0.0002729149,0.005168121,0.002284473,0.0006581329,0.000003456267,0.00002810573,0.0005913868],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3410383,"threshold_uncertainty_score":0.9949871,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9170243300131491,"score_gpt":0.7509985645732905,"score_spread":0.1660257654398586,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}