{"id":"W6947871790","doi":"10.48448/tnqn-bc19","title":"A linguistically-motivated evaluation methodology for unraveling model's abilities in reading comprehension tasks","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Reading comprehension; Comprehension; Benchmark (surveying); Intuition; Annotation; Task (project management); Task analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01197616,0.002815945,0.0008541047,0.003300761,0.0007631192,0.002937518,0.001995867,0.002170186,0.002396737],"category_scores_gemma":[0.07597407,0.0006162482,0.001008243,0.001684789,0.001036273,0.00314181,0.002565031,0.002765462,0.001090445],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001076661,"about_ca_system_score_gemma":0.0013001,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00271225,"about_ca_topic_score_gemma":0.005826527,"domain_scores_codex":[0.988398,0.00645555,0.001160268,0.001599763,0.002073427,0.0003130338],"domain_scores_gemma":[0.9433643,0.04033734,0.003810028,0.005930394,0.005633692,0.0009242549],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.002522356,0.004029665,0.115373,0.003647522,0.001707593,0.0007688338,0.004234951,0.1474739,0.1286456,0.01315056,0.02528091,0.5531651],"study_design_scores_gemma":[0.0004300814,0.003081738,0.05971431,0.0003085069,0.0003843373,0.0003615135,0.0009669327,0.8399897,0.06819458,0.01664793,0.00969795,0.0002224173],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4126526,0.000923367,0.5564863,0.001084363,0.0001775588,0.002237791,0.005192335,0.01174676,0.00949884],"genre_scores_gemma":[0.7054327,0.0001531495,0.2812345,0.0003305615,0.00005612149,0.002615314,0.007902807,0.0008376161,0.001437245],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01197616,"threshold_uncertainty_score":0.06333673,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2127492182627536,"score_gpt":0.4438704265680997,"score_spread":0.2311212083053461,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}