{"id":"W3164690045","doi":"10.14778/3457390.3457398","title":"CBench","year":2021,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Topic Modeling","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Question answering; Benchmark (surveying); Benchmarking; Suite; Set (abstract data type); Vocabulary; Information retrieval; Syntax; Graph; Artificial intelligence; Task (project management); Natural language processing; Theoretical computer science; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0001529272,0.00006907214,0.00009359187,0.00002273597,0.00005942242,0.00006378667,0.0008184388,0.00002254666,0.00001386821],"category_scores_gemma":[0.00006304325,0.00004900509,0.00006871177,0.0002452701,0.00001739739,0.0001820324,0.0006900905,0.00007648387,0.000007355087],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003456443,"about_ca_system_score_gemma":0.00004081607,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000102536,"about_ca_topic_score_gemma":7.709086e-7,"domain_scores_codex":[0.9991588,0.000003286374,0.0001627032,0.000225817,0.0002920495,0.0001573202],"domain_scores_gemma":[0.9994954,0.00001380788,0.0000852946,0.0002055943,0.0001648227,0.00003500634],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.000002173103,0.0001143163,0.004663297,0.00008684125,0.00003529016,0.000002215761,0.001586224,0.00008319807,0.1636389,0.8018087,0.003002491,0.02497638],"study_design_scores_gemma":[0.0003269434,0.00002444722,0.001890345,0.00008302659,0.00001121015,0.00004335591,0.0001719399,0.01171413,0.9319593,0.04455502,0.0090687,0.0001515511],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7545701,0.0011547,0.05893999,0.0271963,0.002263501,0.0006213855,0.00000256241,0.0003187075,0.1549328],"genre_scores_gemma":[0.9619998,0.00001625727,0.03655805,0.0003220028,0.00004904496,0.0000109579,7.134994e-8,0.00000396651,0.001039897],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.7683204,"threshold_uncertainty_score":0.1998369,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01596362296791629,"score_gpt":0.2186539506523365,"score_spread":0.2026903276844202,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}