{"id":"W4395073852","doi":"10.2196/49907","title":"Controlling Inputter Variability in Vignette Studies Assessing Web-Based Symptom Checkers: Evaluation of Current Practice and Recommendations for Isolated Accuracy Metrics","year":2024,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Electronic Health Records Systems","field":"Health Professions","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Vignette; Metric (unit); Quality assurance; Outcome (game theory); Medical physics; Computer science; Contrast (vision); Clinical Practice; Quality (philosophy); Data mining; Reliability engineering; Psychology; Medicine; Artificial intelligence; Social psychology; Engineering; Operations management; Mathematics; Physical therapy; Pathology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5526671,0.002194136,0.004390177,0.008669943,0.001816885,0.007007106,0.008443885,0.002254465,0.003123773],"category_scores_gemma":[0.7487054,0.001612432,0.004555891,0.009160183,0.002645999,0.006515504,0.006065578,0.002579098,0.0008479676],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005940326,"about_ca_system_score_gemma":0.007071833,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00499362,"about_ca_topic_score_gemma":0.006484932,"domain_scores_codex":[0.3681381,0.5095184,0.05449437,0.01352779,0.05264611,0.001675153],"domain_scores_gemma":[0.08979929,0.728714,0.07471961,0.04303025,0.06106829,0.002668494],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.006555756,0.002110452,0.3142196,0.01815506,0.005452267,0.0001634965,0.01525164,0.005218255,0.001789028,0.004025172,0.01247135,0.6145878],"study_design_scores_gemma":[0.00441206,0.02152556,0.730144,0.03152826,0.007019061,0.0009627073,0.009513753,0.1120213,0.01359429,0.0115078,0.05641324,0.001358075],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3802232,0.04365111,0.5096452,0.00957398,0.002692638,0.03087304,0.00345648,0.004684321,0.01519995],"genre_scores_gemma":[0.5588006,0.004617473,0.4060393,0.001318469,0.0005352368,0.02559792,0.001653968,0.0005247766,0.000912269],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4473329,"threshold_uncertainty_score":0.5516412,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4003152487825276,"score_gpt":0.6580511305983481,"score_spread":0.2577358818158205,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}