{"id":"W4293797358","doi":"10.48550/arxiv.2208.12924","title":"Quantifying French Document Complexity","year":2022,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Text Readability and Simplification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Contrast (vision); Task (project management); Natural language processing; Range (aeronautics); Linguistic sequence complexity; Measure (data warehouse); Artificial intelligence; Information retrieval; Linguistics; Data mining","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001820794,0.000588808,0.0005979135,0.008575613,0.001204221,0.002778661,0.0004366639,0.0006124096,0.00352064],"category_scores_gemma":[0.02201957,0.0001506042,0.0004760555,0.005994673,0.0008959395,0.002432916,0.001234078,0.0006514358,0.0005811932],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002187289,"about_ca_system_score_gemma":0.001388389,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02112893,"about_ca_topic_score_gemma":0.01989432,"domain_scores_codex":[0.9969977,0.0007892426,0.0002440107,0.0006213575,0.001184279,0.0001634916],"domain_scores_gemma":[0.9837046,0.008488676,0.001737718,0.001482186,0.004125652,0.0004611033],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000807399,0.0002049416,0.1713646,0.001955308,0.0005592813,0.0008042682,0.007092895,0.0510824,0.04119157,0.05129136,0.02351773,0.6501282],"study_design_scores_gemma":[0.0001092496,0.0005504801,0.5578672,0.0004506065,0.0003284983,0.002400962,0.005733575,0.1645076,0.05785771,0.05043356,0.1594215,0.0003391672],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.8132812,0.00310082,0.1382153,0.0009279213,0.0001114702,0.0003278495,0.01551618,0.002029359,0.02648993],"genre_scores_gemma":[0.899513,0.0007607826,0.07697989,0.00008988225,0.000109768,0.0003399903,0.01771746,0.0003105146,0.004178729],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.02112893,"threshold_uncertainty_score":0.04201186,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2211808650180716,"score_gpt":0.2390983731331629,"score_spread":0.01791750811509135,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}