{"id":"W2017628837","doi":"10.1023/b:qure.0000015307.33811.2d","title":"The stability of utility scores: Test–retest reliability and the interpretation of utility scores in elective total hip arthroplasty","year":2004,"lang":"en","type":"article","venue":"Quality of Life Research","topic":"Health Systems, Economic Evaluations, Quality of Life","field":"Economics, Econometrics and Finance","cited_by":27,"is_retracted":false,"has_abstract":false,"ca_institutions":"Western University; Institute of Health Economics","funders":"","keywords":"Reliability (semiconductor); Physical therapy; Quality of Life Research; Medicine; Test (biology); Arthroplasty; Public health; Surgery; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03462576,0.0005484638,0.0008121495,0.001939247,0.0007533722,0.00181283,0.001217472,0.001246975,0.000490841],"category_scores_gemma":[0.1977136,0.0006444434,0.001467334,0.002128939,0.001704025,0.001622465,0.001439893,0.002119845,0.0003070544],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001122978,"about_ca_system_score_gemma":0.0009205293,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005868186,"about_ca_topic_score_gemma":0.00785341,"domain_scores_codex":[0.9780619,0.01233065,0.002832864,0.001472991,0.0049121,0.0003895153],"domain_scores_gemma":[0.8440853,0.1105726,0.01716059,0.01129948,0.01578822,0.001093876],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001185036,0.0002212035,0.9633641,0.00009412513,0.0008978953,0.00008853004,0.001922006,0.001439783,0.0005776995,0.0004268481,0.0009604806,0.02882223],"study_design_scores_gemma":[0.00002599183,0.000453801,0.9932324,0.00003635381,0.0002143381,0.0002854207,0.0002820774,0.003543271,0.0007725366,0.0005343196,0.0005814905,0.00003819659],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9912454,0.002275946,0.003841714,0.0002046291,0.0001488329,0.00007565322,0.0002540006,0.00002581782,0.001927862],"genre_scores_gemma":[0.9981428,0.0002139794,0.001058518,0.00004442989,0.00004312486,0.00003641524,0.0002439278,0.00001992966,0.0001968683],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9653742,"threshold_uncertainty_score":0.1831207,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3876680495989862,"score_gpt":0.4787481850974712,"score_spread":0.09108013549848504,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}