{"id":"W7084138470","doi":"10.48550/arxiv.2509.25664","title":"QFrBLiMP: a Quebec-French Benchmark of Linguistic Minimal Pairs","year":2025,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Agriculture, Water, and Health","field":"Environmental Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Benchmark (surveying); Hierarchy; Sentence; Statistical model; Computational linguistics; Competence (human resources); Phrase","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0002052843,0.0003208858,0.0004266909,0.0001111924,0.0001412562,0.00002002761,0.0008159773,0.0003648167,0.0009052231],"category_scores_gemma":[0.00005624696,0.0003051123,0.0002299852,0.0004504234,0.0002824054,0.00005913517,0.001363223,0.0005091593,0.0001336387],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000503276,"about_ca_system_score_gemma":0.0001605396,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03240888,"about_ca_topic_score_gemma":0.01360187,"domain_scores_codex":[0.998179,0.000112529,0.0002992808,0.0008659561,0.0001351587,0.0004080917],"domain_scores_gemma":[0.998762,0.00009106017,0.0002864203,0.000616033,0.00004367123,0.0002008198],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.0002494151,0.001201498,0.7914808,0.001621281,0.0003555421,0.000586734,0.00777995,0.1078966,0.0002538938,0.04320155,0.04388359,0.001489042],"study_design_scores_gemma":[0.002330327,0.0004783821,0.8266833,0.001059147,0.0009994026,0.00000914646,0.001391275,0.02940274,0.0005787571,0.09899148,0.03546697,0.00260909],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9227728,0.00007575592,0.001169851,0.00005953899,0.0005483073,0.0004038553,0.00009903894,0.00006285808,0.07480805],"genre_scores_gemma":[0.9791895,0.0002359507,0.0003372887,0.00008328207,0.00009132856,0.000001163813,0.00005548679,0.000008541517,0.01999745],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0784939,"threshold_uncertainty_score":0.9999401,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0324122892452502,"score_gpt":0.1818366409700944,"score_spread":0.1494243517248442,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}