{"id":"W2805879805","doi":"","title":"The 2016 TAC KBP BeSt Evaluation.","year":2016,"lang":"en","type":"article","venue":"Theory and applications of categories","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"","funders":"","keywords":"Computer science","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001073026,0.00004968967,0.00005393439,0.00002048973,0.0002311494,0.0000371375,0.0004178435,0.00002005874,0.00001040658],"category_scores_gemma":[0.00005803744,0.00002554539,0.00001401789,0.00008702723,0.0002028076,0.0001804906,0.00009362456,0.00002308445,0.00002825258],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000007698347,"about_ca_system_score_gemma":0.00005958824,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000004241217,"about_ca_topic_score_gemma":0.000001716809,"domain_scores_codex":[0.9994411,0.00007949142,0.000131713,0.0001369956,0.0001246568,0.00008605378],"domain_scores_gemma":[0.9987498,0.0005063728,0.00006407993,0.0005187953,0.000136011,0.00002497253],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.000001773995,0.000004756812,0.00001290594,0.000001357765,0.000003078323,8.141966e-9,0.00008110808,0.000002581283,0.0003585702,0.6691884,0.00004907178,0.3302964],"study_design_scores_gemma":[0.00009958532,0.00001283613,0.00006855778,0.000005535827,0.000008241096,0.000001614168,0.0001184515,0.0003628192,0.003353573,0.9696741,0.0262462,0.00004854727],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.004620042,0.001279041,0.9866607,0.00173128,0.00004230025,0.0002090352,0.000002443473,0.00003071039,0.005424503],"genre_scores_gemma":[0.9959265,0.0002468295,0.001590516,0.00003195059,0.00005863338,0.000209145,5.659707e-7,0.000002843657,0.001933029],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9913064,"threshold_uncertainty_score":0.1777838,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01721926072958735,"score_gpt":0.2739869869615991,"score_spread":0.2567677262320118,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}