{"id":"W4393335480","doi":"10.1038/s41746-024-01074-z","title":"Foundation metrics for evaluating effectiveness of healthcare conversations powered by generative AI","year":2024,"lang":"en","type":"article","venue":"npj Digital Medicine","topic":"Topic Modeling","field":"Computer Science","cited_by":184,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"National Institute of Standards and Technology","keywords":"Health care; Computer science; Personalization; Set (abstract data type); Process (computing); Comprehension; Human–computer interaction; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02536408,0.00236783,0.001361112,0.006108721,0.001080501,0.003921134,0.001908353,0.002132912,0.002664477],"category_scores_gemma":[0.1398605,0.0004638364,0.001350465,0.003051155,0.001628626,0.0033382,0.002929462,0.001657405,0.0009015829],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001969014,"about_ca_system_score_gemma":0.001665528,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002596543,"about_ca_topic_score_gemma":0.001978738,"domain_scores_codex":[0.9597669,0.02222082,0.005210379,0.002598881,0.009133103,0.001069986],"domain_scores_gemma":[0.813171,0.1479482,0.01168648,0.009905272,0.0149328,0.002356254],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.003525903,0.001894044,0.08912361,0.005973418,0.001468794,0.0002911653,0.004010797,0.1595635,0.0285413,0.01869049,0.008830114,0.6780869],"study_design_scores_gemma":[0.0002811993,0.008378941,0.08310831,0.001364425,0.000888885,0.0008260967,0.002952256,0.7992863,0.053501,0.03228115,0.01662497,0.0005064644],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4173549,0.008741854,0.5330786,0.001433813,0.0005247866,0.004172828,0.005899128,0.004469496,0.02432455],"genre_scores_gemma":[0.7842695,0.0008297157,0.2059418,0.0002527768,0.0001197751,0.003312758,0.003679437,0.0003346324,0.001259737],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02536408,"threshold_uncertainty_score":0.1341397,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05987682264453834,"score_gpt":0.3823261757586979,"score_spread":0.3224493531141596,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}