{"id":"W6891716797","doi":"10.48448/pa3r-x233","title":"Ensuring Safe and High-Quality Outputs: A Guideline Library Approach for Language Models","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Process (computing); Guideline; Language model; Risk assessment; Risk management","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00480511,0.001744829,0.0008892535,0.001693769,0.0006952414,0.00293179,0.003311905,0.002146412,0.006226642],"category_scores_gemma":[0.0245546,0.0009542839,0.001829724,0.001225508,0.001098786,0.003817359,0.003206436,0.003177594,0.005915122],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001486857,"about_ca_system_score_gemma":0.004284252,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005818089,"about_ca_topic_score_gemma":0.01440961,"domain_scores_codex":[0.995267,0.002319341,0.0003894677,0.0007976252,0.001062332,0.0001641638],"domain_scores_gemma":[0.9929959,0.003011543,0.0003934196,0.00207257,0.001346424,0.0001801371],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0002357453,0.0002988357,0.003870266,0.0008982756,0.0002570486,0.000549876,0.0008442359,0.2210465,0.01432975,0.04274937,0.05210187,0.6628181],"study_design_scores_gemma":[0.00007689866,0.0001403594,0.0003342085,0.0001757822,0.00008228446,0.0001694982,0.0001730494,0.8994895,0.0132091,0.05444093,0.03164957,0.00005870032],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.005315088,0.0003517618,0.9668351,0.00074327,0.00005676622,0.0002962363,0.001012366,0.02240161,0.002987662],"genre_scores_gemma":[0.1139773,0.0003768553,0.8727936,0.0009806853,0.0000555031,0.0006573702,0.004641856,0.002931183,0.003585645],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006226642,"threshold_uncertainty_score":0.02541214,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04941460714078977,"score_gpt":0.3340625068712312,"score_spread":0.2846478997304414,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}