{"id":"W4318614900","doi":"10.2196/41516","title":"Calibrating a Transformer-Based Model’s Confidence on Community-Engaged Research Studies: Decision Support Evaluation Study","year":2023,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Advancing Translational Sciences; Center for Clinical and Translational Research; National Institutes of Health; Virginia Commonwealth University","keywords":"Overfitting; Computer science; Artificial intelligence; Machine learning; Transformer; Data science; Artificial neural network; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","sts","research_integrity","insufficient_payload"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1689088,0.0002909657,0.0004188552,0.002277793,0.01076792,0.0007196784,0.002986693,0.0001362438,0.0000544204],"category_scores_gemma":[0.01330186,0.0002511376,0.0001106638,0.007474969,0.0005861284,0.002429139,0.0008124053,0.008354225,0.00154151],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009394163,"about_ca_system_score_gemma":0.001307508,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002921337,"about_ca_topic_score_gemma":0.0007643828,"domain_scores_codex":[0.9489091,0.04007911,0.0009841166,0.0007647156,0.007497472,0.001765511],"domain_scores_gemma":[0.9600053,0.03351172,0.0001156841,0.001747351,0.004338353,0.0002815997],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003809966,0.002184825,0.0001444908,0.0002217301,0.00009492497,0.00007948963,0.8120146,0.06028831,0.002433788,0.01783825,0.01527993,0.08903867],"study_design_scores_gemma":[0.0005656481,0.003317495,0.0003905553,0.0002126561,0.000002925641,0.000001312914,0.2106025,0.7385927,0.01143844,0.03459576,0.00007560539,0.0002044795],"study_design_candidate":"qualitative","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8697135,0.00002997705,0.1187334,0.00166943,0.0002407783,0.004761003,0.00001362662,0.0004262454,0.004412026],"genre_scores_gemma":[0.9964493,0.00003274562,0.0009440446,0.0001008936,0.00003066954,0.00220112,0.00001756401,0.00003212582,0.0001915428],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6783043,"threshold_uncertainty_score":0.9999941,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6147354217783509,"score_gpt":0.5846268103837878,"score_spread":0.03010861139456311,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}