{"id":"W6891761735","doi":"10.48448/4th7-fj78","title":"Generation, Distillation and Evaluation of Motivational Interviewing-Style Reflections with a Foundational Language Model","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Classifier (UML); Language model; Set (abstract data type); Task (project management); Quality (philosophy); Reliability (semiconductor); Language understanding","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001855251,0.0002468947,0.0002159546,0.001265484,0.0001580884,0.0001733762,0.0001972384,0.000107564,0.0006787683],"category_scores_gemma":[0.0002717239,0.0002092281,0.00003064675,0.001351223,0.0007959989,0.000380049,0.00009341003,0.0001685782,0.0001642659],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000443417,"about_ca_system_score_gemma":0.001367592,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001810143,"about_ca_topic_score_gemma":0.003634588,"domain_scores_codex":[0.9967532,0.00008277602,0.0003648538,0.000676996,0.001921498,0.0002006571],"domain_scores_gemma":[0.9982801,0.00002826647,0.0003694108,0.0003304527,0.0009137234,0.00007805003],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00008835003,0.0008763787,0.001313536,0.0008950035,0.0008115587,0.00000499307,0.01372944,0.2790265,0.1792008,0.1698481,0.3118763,0.04232918],"study_design_scores_gemma":[0.0003739877,0.00005484707,0.0001857875,0.0002616807,0.0002119198,0.00001476758,0.0001979375,0.9939775,0.000242081,0.002244602,0.001990375,0.0002445466],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.03907403,0.003655006,0.3222401,0.00105039,0.001252566,0.003660807,0.002203629,0.001136202,0.6257272],"genre_scores_gemma":[0.8942738,0.0000130434,0.04107172,0.00005582794,0.0005495963,0.0001050544,0.001386042,0.0004259896,0.06211898],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8551997,"threshold_uncertainty_score":0.8532076,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1500237487525113,"score_gpt":0.416136112190019,"score_spread":0.2661123634375077,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}