{"id":"W2017628837","doi":"10.1023/b:qure.0000015307.33811.2d","title":"The stability of utility scores: Test–retest reliability and the interpretation of utility scores in elective total hip arthroplasty","year":2004,"lang":"en","type":"article","venue":"Quality of Life Research","topic":"Health Systems, Economic Evaluations, Quality of Life","field":"Economics, Econometrics and Finance","cited_by":27,"is_retracted":false,"has_abstract":false,"ca_institutions":"Western University; Institute of Health Economics","funders":"","keywords":"Reliability (semiconductor); Physical therapy; Quality of Life Research; Medicine; Test (biology); Arthroplasty; Public health; Surgery; Pathology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","sts"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1629128,0.000232968,0.001649136,0.0002909041,0.000350677,0.00005690359,0.0005718261,0.0002170047,0.00009984327],"category_scores_gemma":[0.1666804,0.0001926572,0.0002441024,0.0008262448,0.004786031,0.0004295899,0.0002957293,0.0007736155,0.00002298025],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005443902,"about_ca_system_score_gemma":0.001199071,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02585858,"about_ca_topic_score_gemma":0.005657606,"domain_scores_codex":[0.9835045,0.006573964,0.007633965,0.0009160935,0.000678656,0.0006927908],"domain_scores_gemma":[0.9467915,0.04649403,0.003373353,0.001917825,0.001168536,0.000254743],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.001302237,0.0006292347,0.9265888,0.001563526,0.00006558751,1.084165e-7,0.01218571,0.0001548617,0.00002607817,0.05677497,0.00009675148,0.0006121485],"study_design_scores_gemma":[0.002360816,0.0002723186,0.8469825,0.0001676916,0.000004045383,6.116853e-7,0.00569095,0.005239619,0.0002621511,0.1388235,0.00004235975,0.0001533804],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.982275,0.002239622,0.0008396072,0.01118217,0.0001250062,0.002105416,0.0004475303,0.00001532059,0.0007702943],"genre_scores_gemma":[0.9990403,0.0002831678,0.0003455543,0.0001143231,0.00004803309,0.0001279686,0.00001019702,0.00001647447,0.00001397002],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.08204854,"threshold_uncertainty_score":0.9979224,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3876680495989862,"score_gpt":0.4787481850974712,"score_spread":0.09108013549848504,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}