{"id":"W4390215202","doi":"10.48550/arxiv.2312.14769","title":"Large Language Model (LLM) Bias Index -- LLMBI","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University Canada West","funders":"","keywords":"Metric (unit); Computer science; Index (typography); Measure (data warehouse); Empirical measure; Reliability (semiconductor); Econometrics; Data science; Gender bias; Performance metric; Artificial intelligence; Cognitive psychology; Machine learning; Natural language processing; Psychology; Data mining; Statistics; Social psychology; Mathematics; Economics; Operations management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03188123,0.001661196,0.001521071,0.003945916,0.001422511,0.004782772,0.001980389,0.001923676,0.004789084],"category_scores_gemma":[0.148576,0.000455421,0.001668807,0.003973681,0.001467556,0.005999019,0.004597037,0.003245761,0.00248592],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002333146,"about_ca_system_score_gemma":0.002719965,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002827104,"about_ca_topic_score_gemma":0.003625871,"domain_scores_codex":[0.9701738,0.01505234,0.003014646,0.003113424,0.007884259,0.0007615739],"domain_scores_gemma":[0.8880209,0.08033445,0.007781233,0.01102419,0.01183407,0.001005108],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001927234,0.0004414953,0.1372669,0.003406795,0.002260756,0.0006810941,0.004946992,0.1372253,0.01488943,0.07433155,0.06155361,0.5610688],"study_design_scores_gemma":[0.0001614432,0.0008512241,0.0346226,0.0008818707,0.0006521472,0.001005939,0.002287994,0.6943116,0.02021715,0.1799632,0.06453805,0.0005066954],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1281592,0.002917048,0.8216779,0.00386867,0.0007848808,0.001489278,0.01447315,0.007240308,0.01938953],"genre_scores_gemma":[0.7039112,0.0007463274,0.2742585,0.001477987,0.0003757199,0.002117119,0.0124595,0.001518579,0.003134958],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.03188123,"threshold_uncertainty_score":0.168606,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1565319545738375,"score_gpt":0.2146151699651807,"score_spread":0.05808321539134317,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}