{"id":"W3138215734","doi":"10.33011/computel.v2i.973","title":"Translating Fieldwork into Datasets: The Development of a Corpus for the Quantitative Investigation of Grammatical Phenomena in Eibela","year":2021,"lang":"en","type":"article","venue":"","topic":"Phonetics and Phonology Research","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Annotation; Natural language processing; Realization (probability); Python (programming language); Artificial intelligence; Linguistics; XML; Argument (complex analysis); Process (computing); Programming language; World Wide Web","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006373262,0.0003834011,0.0003756221,0.004331881,0.00217607,0.001729381,0.001581029,0.0009653736,0.01387959],"category_scores_gemma":[0.01699399,0.0005677092,0.0003027234,0.004970966,0.001762628,0.002006054,0.003880645,0.001512944,0.004416935],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001470528,"about_ca_system_score_gemma":0.003871647,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008174762,"about_ca_topic_score_gemma":0.0188223,"domain_scores_codex":[0.9964132,0.001702673,0.0004499547,0.0007635837,0.0005037675,0.0001669537],"domain_scores_gemma":[0.9752969,0.01205934,0.001540955,0.006613666,0.003771517,0.0007176475],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0008043132,0.001447483,0.06455671,0.005473457,0.0001054342,0.002401275,0.1172371,0.003778651,0.08997406,0.05368228,0.1054064,0.5551329],"study_design_scores_gemma":[0.0002218108,0.0002925006,0.2153423,0.001475625,0.00007355544,0.001179282,0.05831178,0.004391314,0.03102242,0.02056461,0.6668856,0.0002391819],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.419365,0.0007563654,0.1807884,0.001909367,0.000510385,0.008864648,0.3198823,0.003326928,0.06459674],"genre_scores_gemma":[0.3766428,0.0003671539,0.3746121,0.0005687193,0.000125983,0.02633752,0.2086868,0.002000194,0.01065873],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01387959,"threshold_uncertainty_score":0.0464319,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1212040605510594,"score_gpt":0.4041265217928668,"score_spread":0.2829224612418074,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}