{"id":"W4409740531","doi":"10.1007/s10994-025-06767-4","title":"Developing safe and responsible large language model: can we balance bias reduction and language understanding?","year":2025,"lang":"en","type":"article","venue":"Machine Learning","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto; Vector Institute","funders":"","keywords":"Reduction (mathematics); Balance (ability); Computer science; Cognitive psychology; Linguistics; Psychology; Mathematics; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009263448,0.001075195,0.001673326,0.0007821863,0.0009173367,0.003342696,0.00352856,0.00222246,0.004007022],"category_scores_gemma":[0.0562905,0.001102831,0.001113168,0.0005248857,0.002567035,0.01588676,0.005320902,0.006344907,0.002510552],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001427475,"about_ca_system_score_gemma":0.005531244,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004369241,"about_ca_topic_score_gemma":0.008204532,"domain_scores_codex":[0.9958795,0.002083718,0.0002237613,0.0006588035,0.0008329463,0.0003211924],"domain_scores_gemma":[0.9679933,0.01729923,0.001373969,0.008687766,0.003674894,0.0009707569],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001063506,0.0004787907,0.01281584,0.0007324649,0.0005327993,0.0004609182,0.001927064,0.165463,0.0274318,0.2581331,0.02272366,0.5082371],"study_design_scores_gemma":[0.00007123703,0.00006191406,0.0003930196,0.00005402486,0.00009104424,0.0001235694,0.0002482494,0.6054031,0.01189956,0.3764156,0.005200272,0.00003841707],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01865494,0.0002800558,0.9720076,0.004459891,0.00009024163,0.00008322762,0.0001746736,0.002832931,0.001416483],"genre_scores_gemma":[0.4906309,0.0006910713,0.4989818,0.002382956,0.0002823063,0.0002900151,0.0008841516,0.002448435,0.003408367],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009263448,"threshold_uncertainty_score":0.04899037,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02426265180724371,"score_gpt":0.3020835596657921,"score_spread":0.2778209078585484,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}