{"id":"W4385570381","doi":"10.18653/v1/2023.acl-long.841","title":"The KITMUS Test: Evaluating Knowledge Integration from Multiple Sources","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University; Mila - Quebec Artificial Intelligence Institute","funders":"Canadian Institute for Advanced Research; Nvidia; Microsoft Research","keywords":"Test (biology); Computer science; Computational linguistics; Artificial intelligence; Geology","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003579584,0.001935732,0.001069554,0.004310282,0.0007776275,0.001978471,0.002676517,0.002371136,0.007471866],"category_scores_gemma":[0.02245249,0.0004572488,0.001047068,0.002572145,0.0005770557,0.004410886,0.004317239,0.001140061,0.003271483],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005902736,"about_ca_system_score_gemma":0.0009052843,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005366194,"about_ca_topic_score_gemma":0.007478705,"domain_scores_codex":[0.9963588,0.001237311,0.0004199602,0.0006943191,0.001176002,0.0001135483],"domain_scores_gemma":[0.9891596,0.007842279,0.0005777964,0.0008364671,0.001085693,0.0004981354],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.01266801,0.002898524,0.05833944,0.005355627,0.006803755,0.002373942,0.001589536,0.02126084,0.0124397,0.002635105,0.06496609,0.8086694],"study_design_scores_gemma":[0.010732,0.01670133,0.242114,0.003079443,0.009795975,0.01090496,0.00758893,0.4771094,0.07235824,0.02581028,0.1228224,0.0009830658],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8712177,0.01791728,0.04565813,0.001831589,0.001076692,0.001729607,0.02021617,0.01125774,0.02909523],"genre_scores_gemma":[0.8259636,0.002841755,0.08912022,0.0006233603,0.000257407,0.001052896,0.06591149,0.0008816143,0.01334764],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007471866,"threshold_uncertainty_score":0.02499586,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03977971996361721,"score_gpt":0.3397854355624773,"score_spread":0.3000057155988601,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}