{"id":"W1656413649","doi":"10.5430/jbgc.v5n2p1","title":"Evaluating agreement between solid tumor measurements used to assess response","year":2015,"lang":"en","type":"article","venue":"Journal of Biomedical Graphics and Computing","topic":"Radiomics and Machine Learning in Medical Imaging","field":"Medicine","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Cancer Institute; Memorial Sloan-Kettering Cancer Center","keywords":"Limits of agreement; Mean difference; Bland–Altman plot; Statistics; Significant difference; Plot (graphics); Distribution (mathematics); Mathematics; Nuclear medicine; Medicine; Mathematical analysis; Confidence interval","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1166632,0.002021828,0.00197107,0.004973885,0.001204316,0.004497423,0.002135669,0.002153639,0.001874703],"category_scores_gemma":[0.256164,0.0007244342,0.001858537,0.003371517,0.003013963,0.003855602,0.002758706,0.001953196,0.0007767736],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001864884,"about_ca_system_score_gemma":0.002188691,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002065186,"about_ca_topic_score_gemma":0.002609752,"domain_scores_codex":[0.862919,0.08281144,0.01327062,0.01081848,0.02891408,0.00126634],"domain_scores_gemma":[0.6921475,0.2113536,0.03391118,0.02268424,0.03856525,0.001338186],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00878961,0.0007466901,0.2953867,0.00437542,0.007512071,0.0004540551,0.007072977,0.04941231,0.03714336,0.01945378,0.01128319,0.5583699],"study_design_scores_gemma":[0.0008313886,0.005845353,0.436278,0.002332642,0.003290047,0.002859841,0.00335946,0.3248402,0.112294,0.06632362,0.04041791,0.001327532],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1325461,0.008607097,0.845175,0.001053898,0.0007856921,0.0009715725,0.001165161,0.003450383,0.006245042],"genre_scores_gemma":[0.7306781,0.00107097,0.2641643,0.0004759954,0.0002011106,0.001189007,0.0007888635,0.0006803913,0.0007512643],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.1166632,"threshold_uncertainty_score":0.6169814,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2153681260112936,"score_gpt":0.4435325674470571,"score_spread":0.2281644414357635,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}