{"id":"W4399678864","doi":"10.1075/dt.24006.lom","title":"The rise of large language models informed by not so large corpora of training data","year":2024,"lang":"en","type":"article","venue":"Digital Translation","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Canadian Standards Association","funders":"","keywords":"Training (meteorology); Computer science; Training set; Natural language processing; Artificial intelligence; Linguistics; Geography; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003785105,0.0000897819,0.0001140319,0.00006311225,0.00005885963,0.0003238552,0.0008417991,0.00005291482,0.000001417009],"category_scores_gemma":[0.00004655194,0.00006421156,0.00004340633,0.0003120777,0.00005034844,0.002869222,0.0001065475,0.0001110768,0.000001522514],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000009022042,"about_ca_system_score_gemma":0.0001058743,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00000767968,"about_ca_topic_score_gemma":0.00001241458,"domain_scores_codex":[0.9990623,0.00001462403,0.0002762334,0.00019778,0.000279469,0.0001696039],"domain_scores_gemma":[0.999195,0.0001870319,0.00008901144,0.0004608854,0.00004230605,0.00002572053],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001966235,0.00004654731,0.00001879663,0.0001918724,0.0000310485,0.000006275758,0.01581812,0.00001228901,0.004654014,0.09662844,0.0006664651,0.8819064],"study_design_scores_gemma":[0.0003796063,0.00006216503,0.00001069811,0.0002759644,0.0000151356,0.000007003722,0.0004072253,0.9234211,0.02588123,0.03601252,0.01330528,0.0002220911],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005352691,0.01090966,0.9807206,0.0003190688,0.0000753866,0.0001652994,0.0008954444,0.0003125272,0.001249361],"genre_scores_gemma":[0.9765292,0.00002816792,0.02313734,0.00002548274,0.0000136253,0.000003364425,0.0002053001,0.000009102856,0.00004842233],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9711765,"threshold_uncertainty_score":0.3122943,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05178205753511441,"score_gpt":0.3195388255157452,"score_spread":0.2677567679806308,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}