{"id":"W4396821301","doi":"10.48550/arxiv.2404.18923","title":"Holmes: A Benchmark to Assess the Linguistic Competence of Language Models","year":2024,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"University of Oxford; Institute for Catastrophic Loss Reduction","keywords":"Benchmark (surveying); Linguistics; Linguistic competence; Competence (human resources); Computer science; Linguistic description; Natural language processing; Psychology; Geography; Philosophy; Social psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007245083,0.002626044,0.000981695,0.005937257,0.001009921,0.00413951,0.002813656,0.002577884,0.005835937],"category_scores_gemma":[0.05195287,0.0006249839,0.001631376,0.003279646,0.00100516,0.006634511,0.004063488,0.002154954,0.005362447],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00129068,"about_ca_system_score_gemma":0.002251081,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006666603,"about_ca_topic_score_gemma":0.009132595,"domain_scores_codex":[0.9913183,0.003503413,0.001320453,0.001297439,0.002147895,0.0004123756],"domain_scores_gemma":[0.9746766,0.01547748,0.0009186522,0.004946144,0.003266589,0.0007145347],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00212149,0.0010458,0.05086058,0.004622029,0.001445382,0.0005049441,0.001647985,0.07747753,0.01329386,0.01936986,0.2710224,0.5565882],"study_design_scores_gemma":[0.0006379615,0.001635279,0.02570843,0.0007240925,0.0004035078,0.0008824917,0.001778215,0.7201078,0.02686776,0.07142711,0.1495028,0.0003245769],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.327642,0.012262,0.3459581,0.003378487,0.001944426,0.00162067,0.1202935,0.1368412,0.0500596],"genre_scores_gemma":[0.5009496,0.001451585,0.2866066,0.001020759,0.0002510662,0.001324741,0.1949783,0.005700398,0.007717038],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.007245083,"threshold_uncertainty_score":0.03831607,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05603746822224436,"score_gpt":0.3264610893305728,"score_spread":0.2704236211083284,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}