{"id":"W4396561228","doi":"10.2196/57978","title":"The Evaluation of Generative AI Should Include Repetition to Assess Stability","year":2024,"lang":"en","type":"article","venue":"JMIR mhealth and uhealth","topic":"Artificial Intelligence in Healthcare and Education","field":"Medicine","cited_by":31,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Generative grammar; Credibility; Computer science; Robustness (evolution); Repetition (rhetorical device); Reliability (semiconductor); Artificial intelligence; Field (mathematics); Randomness; Machine learning","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1901403,0.001469167,0.001994681,0.003032574,0.002524345,0.009652026,0.003936622,0.004608003,0.01047754],"category_scores_gemma":[0.6366693,0.0009727863,0.004009651,0.002541007,0.00651912,0.0113263,0.004435213,0.005019073,0.002526535],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005265248,"about_ca_system_score_gemma":0.005596659,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003823846,"about_ca_topic_score_gemma":0.004497771,"domain_scores_codex":[0.7938352,0.1415644,0.01695599,0.008163216,0.03802806,0.001453184],"domain_scores_gemma":[0.2727708,0.5964332,0.03351055,0.04289509,0.05234348,0.002046917],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.004819187,0.001266036,0.08860248,0.01378542,0.003633563,0.000516398,0.02925862,0.03610644,0.009100444,0.1351435,0.0278426,0.6499253],"study_design_scores_gemma":[0.001024169,0.009131774,0.1183893,0.01444795,0.003016389,0.001421447,0.01225418,0.1199003,0.02284177,0.5469879,0.1485131,0.002071748],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1751032,0.0109529,0.6708264,0.02879561,0.004270955,0.01146602,0.003131371,0.003574805,0.09187876],"genre_scores_gemma":[0.7000887,0.00186611,0.2686394,0.007415259,0.0008070478,0.01433316,0.00124236,0.0008442231,0.004763694],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8098597,"threshold_uncertainty_score":0.9987012,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5899611891369959,"score_gpt":0.599227753469331,"score_spread":0.009266564332335081,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}