{"id":"W4312215389","doi":"10.5334/johd.94","title":"The TRANSCOMP Dataset of Literary Translations from 120 Languages and a Parallel Collection of English-language Originals","year":2022,"lang":"en","type":"article","venue":"Journal of Open Humanities Data","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Linguistics; Computer science; Word (group theory); Metadata; Literary translation; Natural language processing; Literature; History; Art; World Wide Web; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.00140936,0.001549824,0.001075152,0.009182373,0.001480154,0.002151818,0.001923903,0.001633157,0.02525755],"category_scores_gemma":[0.007009558,0.0004069777,0.001037453,0.01015758,0.001011453,0.001837809,0.003228265,0.001572987,0.03728485],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008320847,"about_ca_system_score_gemma":0.00178224,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005032044,"about_ca_topic_score_gemma":0.01312874,"domain_scores_codex":[0.998027,0.0004092733,0.0003451486,0.0003999996,0.0006134851,0.0002050759],"domain_scores_gemma":[0.9951062,0.001338028,0.000352719,0.001203313,0.001528825,0.0004709347],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004636926,0.0003879423,0.005401305,0.002717291,0.00008923571,0.0009803771,0.001102062,0.0006481677,0.004231126,0.002372569,0.9373074,0.04429876],"study_design_scores_gemma":[0.0002794582,0.0001190445,0.02242036,0.0004726137,0.00005043485,0.001068203,0.001839859,0.001273947,0.003441569,0.001405845,0.967553,0.00007570979],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.01692666,0.0006954038,0.001257984,0.0002683159,0.0003063066,0.0001818531,0.9729258,0.001294454,0.006143227],"genre_scores_gemma":[0.00425132,0.0001194938,0.001760456,0.00004961857,0.00005410174,0.0002281405,0.9915614,0.0001475358,0.00182808],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.9980761,"threshold_uncertainty_score":0.08449495,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05700619531125964,"score_gpt":0.3531706120189455,"score_spread":0.2961644167076859,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}