{"id":"W4404782657","doi":"10.18653/v1/2024.emnlp-main.1021","title":"A linguistically-motivated evaluation methodology for unraveling model’s abilities in reading comprehension tasks","year":2024,"lang":"en","type":"article","venue":"","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"Grand Équipement National De Calcul Intensif","keywords":"Computer science; Reading comprehension; Comprehension; Reading (process); Natural language processing; Program comprehension; Artificial intelligence; Cognitive psychology; Linguistics; Psychology; Programming language; Software","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003354611,0.0001275929,0.0002085681,0.0002486355,0.00008637572,0.0001727089,0.0002257712,0.00008756662,0.00001082689],"category_scores_gemma":[0.001432234,0.000109529,0.00006911792,0.0002502072,0.00001857553,0.0001631869,0.00008983228,0.0001826065,0.0000133337],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001671506,"about_ca_system_score_gemma":0.0001331833,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001473796,"about_ca_topic_score_gemma":0.00001142526,"domain_scores_codex":[0.9983286,0.0003018282,0.0003775564,0.0004765058,0.0002349642,0.0002806054],"domain_scores_gemma":[0.9978238,0.001589064,0.00003849671,0.0002109243,0.0002995883,0.00003816185],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00000632857,0.000009672856,0.00003806352,0.00008617841,0.00001118967,0.00000358204,0.002388471,0.1114368,0.007999497,0.8644269,0.00003206807,0.01356126],"study_design_scores_gemma":[0.0001079978,0.0000519555,0.00004372681,0.0002745465,0.000007190341,0.000004998492,0.0001709349,0.9506598,0.002753187,0.04337371,0.00242452,0.0001274587],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01363531,0.0002087987,0.9824517,0.0002209161,0.0009477732,0.0003950496,8.532443e-7,0.0002527275,0.001886874],"genre_scores_gemma":[0.6969491,0.000002639242,0.301123,0.00004986654,0.0001154723,0.00004491788,0.000003505784,0.00001110311,0.001700426],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.839223,"threshold_uncertainty_score":0.4466465,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1879495290308362,"score_gpt":0.3927184321625435,"score_spread":0.2047689031317073,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}