{"id":"W6947908880","doi":"10.48448/090r-bw49","title":"WSC+: Enhancing The Winograd Schema Challenge Using Tree-of-Experts","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Brock University","funders":"","keywords":"Schema (genetic algorithms); Overconfidence effect; Benchmark (surveying); Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007995419,0.001892725,0.0006887955,0.002383498,0.001117191,0.002913462,0.002546692,0.003900004,0.01175317],"category_scores_gemma":[0.03188175,0.0004172241,0.001467875,0.001453663,0.001078794,0.004826674,0.004119324,0.003500044,0.008329572],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00152264,"about_ca_system_score_gemma":0.002643097,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006854647,"about_ca_topic_score_gemma":0.01121304,"domain_scores_codex":[0.9910433,0.005194089,0.0005294955,0.001619569,0.00132628,0.0002872345],"domain_scores_gemma":[0.9801846,0.01277325,0.0007013631,0.003472731,0.00208947,0.0007786361],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001005401,0.001110492,0.01740437,0.002423273,0.0002810685,0.001259504,0.00218456,0.02939484,0.007755302,0.01534871,0.6095869,0.3122456],"study_design_scores_gemma":[0.001005588,0.0007976012,0.01272379,0.0007149399,0.0001558545,0.002015399,0.003279278,0.4823552,0.02815654,0.04576223,0.4227401,0.0002936022],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.372306,0.007613754,0.2256295,0.01693277,0.003653985,0.003672836,0.2195052,0.0749874,0.07569858],"genre_scores_gemma":[0.3813258,0.0008260259,0.2387145,0.003298091,0.0004816152,0.001476151,0.35101,0.003425365,0.0194424],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01175317,"threshold_uncertainty_score":0.04228431,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05551896162194712,"score_gpt":0.343597653360824,"score_spread":0.2880786917388768,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}