{"id":"W3035455711","doi":"10.1007/978-3-030-58323-1_16","title":"Evaluating a Multi-sense Definition Generation Model for Multiple Languages","year":2020,"lang":"en","type":"preprint","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Polysemy; Computer science; Contrast (vision); Word (group theory); Context (archaeology); Natural language processing; Artificial intelligence; Language model; Word-sense disambiguation; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001299258,0.000358636,0.000365822,0.0003274488,0.0002743671,0.0007447305,0.001710029,0.0002354304,6.799405e-7],"category_scores_gemma":[0.001135186,0.0003573185,0.0001282957,0.0005712802,0.0001104607,0.0004313417,0.002210262,0.0005948816,0.000004324469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002409705,"about_ca_system_score_gemma":0.000693164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005698181,"about_ca_topic_score_gemma":0.0001622205,"domain_scores_codex":[0.9962991,0.0001103154,0.0005100481,0.001851598,0.0006878096,0.0005411376],"domain_scores_gemma":[0.9977305,0.0004306754,0.000261197,0.001127106,0.0003207193,0.0001298358],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002922346,0.00002598798,0.00006357756,0.00005401925,0.00000330817,0.000005756943,0.002645629,0.8196637,0.008254548,0.0003615402,0.000002475758,0.1689166],"study_design_scores_gemma":[0.0004694379,0.00005416184,0.00005531885,0.0001048859,0.000006714839,0.00000889509,6.903194e-7,0.9672149,0.006299736,0.02540887,7.296659e-7,0.0003756714],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.03160217,0.0001727414,0.9644215,0.001450869,0.001199206,0.0008839634,0.00001371004,0.0002534505,0.00000235036],"genre_scores_gemma":[0.4833863,0.000003208849,0.5153883,0.0008224472,0.0003046235,0.000070047,0.00001387027,0.00001093374,3.008088e-7],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.4517841,"threshold_uncertainty_score":0.9998879,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2100460893627557,"score_gpt":0.3725867930988983,"score_spread":0.1625407037361426,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}