{"id":"W2092865291","doi":"10.7202/1008341ar","title":"Reliability and Validity of a Scale-based Assessment for Translation Tests","year":2012,"lang":"en","type":"article","venue":"Meta Journal des traducteurs","topic":"Translation Studies and Practices","field":"Arts and Humanities","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Scale (ratio); Reliability (semiconductor); Test (biology); Computer science; Quality (philosophy); Christian ministry; Artificial intelligence; Translation (biology); Natural language processing; Mathematics education; Machine learning; Psychology","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001391768,0.0001088195,0.0002568357,0.0000550865,0.0003841436,0.00009356351,0.00005282237,0.00002266695,0.0002185483],"category_scores_gemma":[0.00003069115,0.00007658118,0.0001847647,0.0000291628,0.0002262562,0.0007028435,0.000002727131,0.0001209095,5.603028e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001224968,"about_ca_system_score_gemma":0.00002328151,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002081599,"about_ca_topic_score_gemma":0.0001096379,"domain_scores_codex":[0.9990635,0.0001788108,0.0003169909,0.00009240377,0.0001752297,0.0001730617],"domain_scores_gemma":[0.9988627,0.0006201871,0.0001948745,0.00008122533,0.0001583571,0.00008261737],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005150191,0.001303654,0.07408097,0.0009829961,0.001451872,8.094152e-7,0.01674015,0.0002218309,0.002334547,0.02473963,0.000836377,0.8767921],"study_design_scores_gemma":[0.001048562,0.0003333689,0.08448731,0.00002457882,0.001882268,0.000011988,0.0003659708,0.0001989115,0.000466521,0.009541484,0.9014464,0.000192617],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.878644,0.05405474,0.05888699,0.002573414,0.001313892,0.0006543101,0.000183623,0.00004516244,0.003643898],"genre_scores_gemma":[0.9828062,0.001498485,0.0152501,0.000050584,0.0003240908,0.0000145012,0.000004016395,0.00001066448,0.00004135782],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.90061,"threshold_uncertainty_score":0.312289,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.165354839471253,"score_gpt":0.3437038458073066,"score_spread":0.1783490063360536,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}