{"id":"W2736098086","doi":"","title":"Community-based corpus-building: Three case studies","year":2017,"lang":"en","type":"article","venue":"The COCOON platform (University of Paris)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Lemmatisation; Linguistics; Corpus linguistics; Annotation; Focus (optics); Ethos; Narrative; Documentation; Artificial intelligence; Natural language processing","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["sts"],"consensus_categories":[],"category_scores_codex":[0.0008351031,0.0001380949,0.0002392212,0.00009041883,0.003017443,0.000125887,0.003416815,0.0000801934,0.000005217115],"category_scores_gemma":[0.0001101245,0.0001140284,0.00007857718,0.0001553951,0.0005892467,0.001010676,0.001546129,0.0004718967,0.000008803985],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006073431,"about_ca_system_score_gemma":0.00006643491,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001574927,"about_ca_topic_score_gemma":0.001218323,"domain_scores_codex":[0.9992515,0.00005503698,0.0001042504,0.0001677341,0.0002134734,0.0002079369],"domain_scores_gemma":[0.9974229,0.0002456174,0.0003844185,0.001645847,0.0002461995,0.00005499604],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004240726,0.0005769171,0.005235475,0.001052726,0.0006965355,0.006944839,0.0353285,0.0001527477,0.006730709,0.6428419,0.01981096,0.2802047],"study_design_scores_gemma":[0.006084818,0.00151397,0.004308649,0.001484282,0.0005202856,0.002466901,0.01363259,0.1427452,0.05002011,0.7590272,0.01570196,0.002494027],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4727626,0.001385958,0.5206815,0.003104895,0.0002090249,0.0003619487,0.000008627827,0.0005157577,0.0009697357],"genre_scores_gemma":[0.8806484,0.00002732048,0.1190631,0.0001072903,0.00001173066,4.182457e-7,0.000001294618,0.000005448208,0.000135007],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.4078858,"threshold_uncertainty_score":0.9982805,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07443160432197264,"score_gpt":0.3087848184024849,"score_spread":0.2343532140805122,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}