{"id":"W2748411071","doi":"10.1017/9781316416013.012","title":"Appendix: <i>Corpora and Text Collections</i>","year":2017,"lang":"en","type":"other","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Appendix; Computer science; Information retrieval; Natural language processing; Biology; Paleontology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.001831771,0.001739821,0.001553468,0.01056777,0.001265072,0.003004284,0.002850429,0.001206298,0.725719],"category_scores_gemma":[0.01731649,0.001077654,0.0008155626,0.01760938,0.000627496,0.003276059,0.002575113,0.001688575,0.628998],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001288431,"about_ca_system_score_gemma":0.003502736,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009120999,"about_ca_topic_score_gemma":0.0104489,"domain_scores_codex":[0.9983486,0.0002760474,0.0002852957,0.0003061823,0.0006600288,0.0001238596],"domain_scores_gemma":[0.9805934,0.005829284,0.0009062854,0.003321541,0.008255041,0.001094343],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001915437,0.00001234491,0.00003513408,0.0002112197,0.000002364332,0.00000957984,0.00000995347,0.00004010988,0.00008993939,0.0004604076,0.9922751,0.006834826],"study_design_scores_gemma":[0.00008292001,0.00001496361,0.0006571199,0.0001722029,0.00001006397,0.00007164163,0.00004299389,0.0002366839,0.000690544,0.002098276,0.9959024,0.00002017641],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"other","genre_scores_codex":[0.000208669,0.0001863638,0.002852762,0.0004197251,0.0005683959,0.0002885664,0.9323753,0.005530853,0.05756949],"genre_scores_gemma":[0.001130848,0.000325154,0.008548921,0.0002378974,0.0002578498,0.0008722389,0.9433371,0.00481105,0.04047908],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.725719,"threshold_uncertainty_score":0.3912285,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01029422964141303,"score_gpt":0.2567647533046196,"score_spread":0.2464705236632065,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}