{"id":"W6948968340","doi":"10.5281/zenodo.10428713","title":"Large Language Model (LLM) Bias Index—LLMBI","year":2023,"lang":"en","type":"other","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Libraries and Information Services","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University Canada West","funders":"","keywords":"Measure (data warehouse); Reliability (semiconductor); Empirical measure; Language model; Process (computing); Diversity (politics); Gender bias","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02917442,0.001681585,0.001496781,0.004216729,0.001475209,0.004569565,0.001881494,0.001908637,0.004494495],"category_scores_gemma":[0.1471416,0.000450471,0.001552193,0.004274038,0.001426251,0.005719583,0.004620277,0.003074195,0.002619697],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002115865,"about_ca_system_score_gemma":0.002501205,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002911857,"about_ca_topic_score_gemma":0.003682431,"domain_scores_codex":[0.9705796,0.01411952,0.003068103,0.003148646,0.00828641,0.0007977908],"domain_scores_gemma":[0.8945759,0.07400014,0.008094577,0.01064681,0.01164536,0.001037218],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001944016,0.0004595404,0.1753204,0.003485534,0.002206067,0.0006848608,0.006136925,0.1105411,0.0165648,0.05994087,0.06609958,0.5566162],"study_design_scores_gemma":[0.0001836336,0.001004062,0.05395567,0.0009733613,0.0007821076,0.001219468,0.003139211,0.6654703,0.02547147,0.1643803,0.08275282,0.0006675172],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1848955,0.003371113,0.7577162,0.003970741,0.0008774824,0.00161092,0.0167888,0.007797987,0.02297131],"genre_scores_gemma":[0.7406188,0.0007214883,0.2373013,0.001392349,0.0003520664,0.002084909,0.01296674,0.001569985,0.00299241],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02917442,"threshold_uncertainty_score":0.1542909,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0665843376175713,"score_gpt":0.2379908573496372,"score_spread":0.1714065197320659,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}