{"id":"W1981789870","doi":"10.1016/j.neuroimage.2014.01.058","title":"Hippocampal volume change measurement: Quantitative assessment of the reproducibility of expert manual outlining and the automated methods FreeSurfer and FIRST","year":2014,"lang":"en","type":"article","venue":"NeuroImage","topic":"Dementia and Cognitive Impairment Research","field":"Medicine","cited_by":130,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"Genentech; National Institutes of Health; Servier; Canadian Institutes of Health Research; Vrije Universiteit Amsterdam; Bayer HealthCare; European Commission; Seventh Framework Programme; Takeda Pharmaceutical Company; GE Healthcare; Abbott Laboratories; BioClinica; Novartis Pharmaceuticals Corporation; Pfizer; Alzheimer's Drug Discovery Foundation; Merck; Amorfix Life Sciences; Eli Lilly and Company; Bristol-Myers Squibb; National Institute on Aging; Alzheimer's Association; F. Hoffmann-La Roche; Foundation for the National Institutes of Health","keywords":"Reproducibility; Neuroimaging; Alzheimer's Disease Neuroimaging Initiative; Hippocampal formation; Atrophy; Brain size; Cognitive impairment; Medicine; Nuclear medicine; Cognition; Magnetic resonance imaging; Pathology; Internal medicine; Radiology; Statistics; Mathematics; Psychiatry","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"},{"model":"gpt","categories":["metaresearch"],"domain":"methods","study_design":"design_other","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.02619305,0.001053288,0.001424081,0.004338524,0.00134859,0.002313892,0.001717853,0.001980205,0.001740814],"category_scores_gemma":[0.05481569,0.001079341,0.001107639,0.00183461,0.00206333,0.001697639,0.001912367,0.0009517038,0.0005327983],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007615443,"about_ca_system_score_gemma":0.001127881,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003342261,"about_ca_topic_score_gemma":0.007873301,"domain_scores_codex":[0.9803736,0.007349093,0.002300801,0.004120856,0.005362339,0.0004933359],"domain_scores_gemma":[0.9464571,0.02314917,0.005099494,0.01322263,0.01143116,0.0006405062],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.01298574,0.0008809625,0.2984542,0.002689908,0.008098176,0.0004590555,0.007314194,0.01416763,0.1526041,0.006250693,0.006008126,0.4900872],"study_design_scores_gemma":[0.0007964341,0.003177636,0.7522001,0.0002698172,0.002501855,0.006445142,0.000930347,0.09263841,0.1247685,0.006272495,0.009386928,0.0006124362],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6981425,0.005752848,0.2817177,0.0002794237,0.0003852441,0.0007731882,0.00174264,0.00356253,0.007643865],"genre_scores_gemma":[0.8833476,0.0005187301,0.1114657,0.0001560333,0.00009024225,0.0004259562,0.0007640726,0.001298192,0.001933467],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.973807,"threshold_uncertainty_score":0.1385238,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1028280826469189,"score_gpt":0.427490044522795,"score_spread":0.3246619618758761,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}