{"id":"W6891716797","doi":"10.48448/pa3r-x233","title":"Ensuring Safe and High-Quality Outputs: A Guideline Library Approach for Language Models","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Process (computing); Guideline; Language model; Risk assessment; Risk management","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.002527533,0.0006614923,0.0007734076,0.001478038,0.0001796334,0.0005940084,0.001181992,0.000360828,0.0002784397],"category_scores_gemma":[0.0002310554,0.0005674824,0.0001246711,0.001327665,0.001265748,0.0007993089,0.000954397,0.0004596349,0.0004567759],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001541338,"about_ca_system_score_gemma":0.0006473971,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0008494481,"about_ca_topic_score_gemma":0.0000895093,"domain_scores_codex":[0.9950993,0.00009298069,0.0008838163,0.001911395,0.001080674,0.0009318087],"domain_scores_gemma":[0.9978851,0.00009810275,0.000423021,0.001116611,0.0001075469,0.0003696277],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00005402388,0.0002207341,0.000009753084,0.001222059,0.0001290479,0.00002490641,0.0008579807,0.001044249,0.004677443,0.09483421,0.8876921,0.009233501],"study_design_scores_gemma":[0.002081264,0.0001876182,0.00001306761,0.0008624762,0.0003166267,0.00005955862,0.001896049,0.5037495,0.00124488,0.0363724,0.4505677,0.002648893],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.0002265407,0.01236343,0.1919545,0.000916537,0.0009030052,0.002748012,0.006686795,0.003465682,0.7807355],"genre_scores_gemma":[0.009848895,0.000057465,0.4890897,0.0005814628,0.001323655,0.0001238598,0.001092372,0.001941735,0.4959408],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.5027052,"threshold_uncertainty_score":0.9996777,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04941460714078977,"score_gpt":0.3340625068712312,"score_spread":0.2846478997304414,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}