{"id":"W2967423807","doi":"","title":"Lemmatising Treebanks. Corpus Annotation with Knowledge Bases","year":2018,"lang":"es","type":"article","venue":"RAEL: revista electrónica de lingüística aplicada","topic":"Lexicography and Language Studies","field":"Arts and Humanities","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Treebank; Annotation; Artificial intelligence; Parsing; Natural language processing; Humanities; Linguistics; Computer science; Corpus linguistics; Art; Philosophy","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005371185,0.0008419582,0.0007341463,0.009263111,0.001964252,0.004827,0.001673661,0.0009747499,0.02236148],"category_scores_gemma":[0.02222618,0.00114225,0.0007619162,0.01407721,0.001465133,0.005605272,0.003596665,0.001935062,0.01357105],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00193085,"about_ca_system_score_gemma":0.004353888,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006439035,"about_ca_topic_score_gemma":0.008127138,"domain_scores_codex":[0.9953811,0.002008724,0.0006297322,0.0009434297,0.0009048118,0.0001322471],"domain_scores_gemma":[0.9869311,0.007198948,0.001102123,0.002630051,0.001971136,0.0001665428],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004008309,0.0001433678,0.002900403,0.003339282,0.0001431609,0.0007359161,0.008083941,0.005301735,0.02261397,0.118167,0.1223073,0.715863],"study_design_scores_gemma":[0.0001215332,0.0000700052,0.007573058,0.001248673,0.0001318971,0.0007242084,0.003707818,0.04207583,0.0336421,0.1235161,0.7870103,0.0001784824],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01259224,0.00180548,0.8938358,0.001377752,0.0004343238,0.00114701,0.0338175,0.02769168,0.02729818],"genre_scores_gemma":[0.06713866,0.001361615,0.8561233,0.0003667594,0.0001587574,0.001656897,0.05516043,0.005632043,0.01240151],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02236148,"threshold_uncertainty_score":0.07480663,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01399181277315071,"score_gpt":0.2642212846142912,"score_spread":0.2502294718411405,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}