{"id":"W4318614900","doi":"10.2196/41516","title":"Calibrating a Transformer-Based Model’s Confidence on Community-Engaged Research Studies: Decision Support Evaluation Study","year":2023,"lang":"en","type":"article","venue":"JMIR Formative Research","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Center for Advancing Translational Sciences; Center for Clinical and Translational Research; National Institutes of Health; Virginia Commonwealth University","keywords":"Overfitting; Computer science; Artificial intelligence; Machine learning; Transformer; Data science; Artificial neural network; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.06238117,0.001388747,0.001009346,0.003856187,0.0007183631,0.004289935,0.00233557,0.002893194,0.001478416],"category_scores_gemma":[0.1919262,0.0004288283,0.001280848,0.002620707,0.001527371,0.003920034,0.002870191,0.002557295,0.0003910176],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002852844,"about_ca_system_score_gemma":0.00184224,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003225273,"about_ca_topic_score_gemma":0.002590346,"domain_scores_codex":[0.974638,0.01661322,0.002249855,0.003012178,0.002854191,0.0006325276],"domain_scores_gemma":[0.7446681,0.2057521,0.01352931,0.01293427,0.01902377,0.004092408],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.007371872,0.002702253,0.3553548,0.001219748,0.0024438,0.0004126995,0.001631327,0.379285,0.00262994,0.008094149,0.00625195,0.2326025],"study_design_scores_gemma":[0.0006097747,0.001901701,0.02998021,0.0002994482,0.0006453656,0.000255844,0.0006754475,0.9507541,0.004429822,0.007792182,0.002542097,0.0001140343],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9084225,0.001780041,0.08133554,0.00138767,0.0001316787,0.0008523106,0.001242379,0.0004444572,0.00440339],"genre_scores_gemma":[0.9746602,0.0001414019,0.02382574,0.0001132168,0.0000257012,0.0001933911,0.0008107011,0.00002225387,0.000207561],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9376189,"threshold_uncertainty_score":0.3299071,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.6147354217783509,"score_gpt":0.5846268103837878,"score_spread":0.03010861139456311,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}