{"id":"W2134830997","doi":"10.7202/029804ar","title":"Constructing a Large-Scale English-Persian Parallel Corpus","year":2009,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Payame Noor University; University of Warwick","keywords":"Persian; Computer science; Natural language processing; Machine translation; Task (project management); Artificial intelligence; The Internet; String (physics); Scale (ratio); Software; Corpus linguistics; Construct (python library); Linguistics; World Wide Web; Information retrieval; Programming language; Engineering; Mathematics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003318689,0.0008249718,0.000718091,0.004458405,0.002283338,0.001511403,0.001453879,0.0007080032,0.01254997],"category_scores_gemma":[0.009396739,0.0006920755,0.0005350173,0.005708892,0.001245219,0.002626512,0.002221045,0.001349599,0.004244595],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008809482,"about_ca_system_score_gemma":0.002363522,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002569223,"about_ca_topic_score_gemma":0.003445937,"domain_scores_codex":[0.9981345,0.0007669111,0.0002331302,0.0004588102,0.0003239508,0.00008264035],"domain_scores_gemma":[0.9925036,0.003331071,0.000295133,0.001192849,0.002463323,0.0002140669],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001510599,0.001305237,0.008328275,0.00334485,0.0001818849,0.008514601,0.01408004,0.01883725,0.07100336,0.06573015,0.08707112,0.7200926],"study_design_scores_gemma":[0.0007388224,0.0009924687,0.02333256,0.0005344414,0.0003773744,0.004933973,0.01075682,0.07077941,0.1136413,0.0450353,0.7286063,0.0002711909],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3490332,0.002326905,0.5380201,0.001656067,0.001044882,0.005625725,0.03196488,0.00889987,0.06142848],"genre_scores_gemma":[0.2480167,0.0008025315,0.675638,0.0002462857,0.0002212027,0.003421723,0.05746055,0.001430988,0.01276205],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01254997,"threshold_uncertainty_score":0.04198384,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01813183722996285,"score_gpt":0.2598321190240259,"score_spread":0.2417002817940631,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}