{"id":"W4317504322","doi":"10.1145/3580495","title":"Filtering and Extended Vocabulary based Translation for Low-resource Language Pair of Sanskrit-Hindi","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Asian and Low-Resource Language Information Processing","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Sanskrit; Computer science; Machine translation; Hindi; Natural language processing; Artificial intelligence; Vocabulary; Transformer; Sentence; Phrase; Language translation; Linguistics; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004744751,0.0006355652,0.0005959225,0.0006878676,0.0005542416,0.0005834477,0.0005359831,0.0004566738,0.002881867],"category_scores_gemma":[0.001241812,0.0001886976,0.0006970569,0.0006522863,0.0003485652,0.000979682,0.0006324204,0.0006009483,0.00166468],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003227034,"about_ca_system_score_gemma":0.0008452961,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004116355,"about_ca_topic_score_gemma":0.007836582,"domain_scores_codex":[0.9996331,0.00008793104,0.00002889885,0.0001351415,0.00007468477,0.00004012702],"domain_scores_gemma":[0.9996064,0.00013712,0.0000290165,0.00007172816,0.0001370273,0.0000186754],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0003994922,0.0002299086,0.00171714,0.0004057113,0.00008049883,0.0006683163,0.0005453193,0.02673599,0.1740093,0.008978787,0.005261214,0.7809682],"study_design_scores_gemma":[0.00006137748,0.0006635025,0.003380689,0.00003925425,0.0001283679,0.001297936,0.0003536165,0.8135803,0.1522638,0.008518704,0.01965353,0.00005891658],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1347155,0.000762505,0.8551548,0.000176168,0.0001967597,0.0001597898,0.0003332829,0.003235278,0.005265936],"genre_scores_gemma":[0.5433568,0.0004870761,0.4411255,0.0001792493,0.00008660204,0.0002136454,0.002611052,0.0003487521,0.01159137],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004116355,"threshold_uncertainty_score":0.009640753,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.00992098405943372,"score_gpt":0.2561712473687812,"score_spread":0.2462502633093475,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}