{"id":"W4404782657","doi":"10.18653/v1/2024.emnlp-main.1021","title":"A linguistically-motivated evaluation methodology for unraveling model’s abilities in reading comprehension tasks","year":2024,"lang":"en","type":"article","venue":"","topic":"Intelligent Tutoring Systems and Adaptive Learning","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Computer Research Institute of Montréal","funders":"Grand Équipement National De Calcul Intensif","keywords":"Computer science; Reading comprehension; Comprehension; Reading (process); Natural language processing; Program comprehension; Artificial intelligence; Cognitive psychology; Linguistics; Psychology; Programming language; Software","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01241936,0.002551132,0.0008859818,0.003302651,0.0007513343,0.00287101,0.001927597,0.002164662,0.001949901],"category_scores_gemma":[0.08896585,0.000587429,0.0009443375,0.001750639,0.001047577,0.003198989,0.002416264,0.002641083,0.0009325298],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009818209,"about_ca_system_score_gemma":0.001135455,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002067566,"about_ca_topic_score_gemma":0.004085735,"domain_scores_codex":[0.9875629,0.006538232,0.001395171,0.001711578,0.002430943,0.0003611782],"domain_scores_gemma":[0.9265057,0.05349758,0.005068815,0.006510475,0.00725924,0.001158194],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002698767,0.004262055,0.1352065,0.003739745,0.001509113,0.0007416062,0.005728329,0.1231775,0.1508161,0.01124283,0.01685622,0.5440213],"study_design_scores_gemma":[0.0004344561,0.004581925,0.07875466,0.0003232658,0.0004358573,0.0004589843,0.00138798,0.7958621,0.09156678,0.01622414,0.00970544,0.0002644818],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4727928,0.0006944101,0.5040327,0.0007921062,0.0001574196,0.002153766,0.00353034,0.008717176,0.007129228],"genre_scores_gemma":[0.7587759,0.0001302203,0.2314456,0.0002433305,0.00005239591,0.002453022,0.005117856,0.0005876702,0.001194008],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01241936,"threshold_uncertainty_score":0.06568062,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1879495290308362,"score_gpt":0.3927184321625435,"score_spread":0.2047689031317073,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}