{"id":"W4399985342","doi":"","title":"Étude des facteurs de complexité des modèles de langage dans une tâche de compréhension de lecture à l'aide d'une expérience contrôlée sémantiquement","year":2024,"lang":"fr","type":"article","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Linguistics and Discourse Analysis","field":"Arts and Humanities","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Psychology; Humanities; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00635842,0.001171535,0.0006307061,0.001505444,0.000996397,0.00775444,0.0007926796,0.002431156,0.00564509],"category_scores_gemma":[0.05242368,0.0010493,0.0008489402,0.000931245,0.002970503,0.01004427,0.001757966,0.003194317,0.000749082],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002402632,"about_ca_system_score_gemma":0.001057871,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01198573,"about_ca_topic_score_gemma":0.004769221,"domain_scores_codex":[0.9928198,0.00312306,0.000429985,0.001418078,0.001969796,0.0002392427],"domain_scores_gemma":[0.9290366,0.06073904,0.002270364,0.002740056,0.004825669,0.0003881584],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"nonrandomized_trial","study_design_scores_codex":[0.002478158,0.0003791346,0.04573192,0.001237256,0.0006285424,0.005279659,0.1889536,0.03804334,0.2435059,0.2287828,0.0042493,0.2407305],"study_design_scores_gemma":[0.000340528,0.0008593232,0.06292178,0.0003927453,0.0008242229,0.004398591,0.03848572,0.6375759,0.1382486,0.07043123,0.04507766,0.0004436319],"study_design_candidate":"nonrandomized_trial","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.522588,0.0005183755,0.4556503,0.001496714,0.0001172698,0.0001791805,0.0004068661,0.002035616,0.01700778],"genre_scores_gemma":[0.9612656,0.0001233695,0.03481015,0.00005208725,0.00002300952,0.00008242975,0.0002369339,0.0003307618,0.003075631],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01198573,"threshold_uncertainty_score":0.03362691,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0365092743478714,"score_gpt":0.2612650919884398,"score_spread":0.2247558176405684,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}