{"id":"W3197016261","doi":"10.14778/3476311.3476326","title":"CBench","year":2021,"lang":"en","type":"article","venue":"Proceedings of the VLDB Endowment","topic":"Topic Modeling","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Benchmarking; Suite; Benchmark (surveying); Question answering; Set (abstract data type); Task (project management); Quality (philosophy); Artificial intelligence; Information retrieval; Natural language processing; Programming language; Systems engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004941433,0.002283475,0.001285439,0.004493848,0.001330225,0.003980914,0.004195515,0.001906422,0.03381145],"category_scores_gemma":[0.02090985,0.0008568113,0.001444134,0.004420688,0.0008460794,0.00486536,0.004248819,0.002517849,0.03198352],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001617888,"about_ca_system_score_gemma":0.002900905,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009404302,"about_ca_topic_score_gemma":0.008403559,"domain_scores_codex":[0.9928378,0.001530305,0.0008526623,0.001199839,0.002968545,0.000610826],"domain_scores_gemma":[0.9865362,0.003396722,0.0005082141,0.003979649,0.005037015,0.0005423375],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001643959,0.0009005594,0.005525374,0.002940087,0.000225936,0.0003418904,0.0009186993,0.01077259,0.0123477,0.02738145,0.6467527,0.290249],"study_design_scores_gemma":[0.0004214914,0.0007154468,0.006729153,0.0005047367,0.0001282518,0.00051455,0.0007034084,0.09204452,0.02749698,0.03417563,0.8363254,0.0002405069],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"software","genre_gemma":"methods","genre_scores_codex":[0.0424864,0.005157063,0.1893052,0.001597542,0.00166571,0.002185362,0.08803118,0.5442054,0.1253662],"genre_scores_gemma":[0.2062019,0.002629763,0.288634,0.002142327,0.0003266974,0.00365202,0.3755449,0.05745689,0.06341153],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.03381145,"threshold_uncertainty_score":0,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01596362296791629,"score_gpt":0.2186539506523365,"score_spread":0.2026903276844202,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}