{"id":"W6929445656","doi":"10.48448/me00-k761","title":"A User-Centric Multi-Intent Benchmark for Evaluating Large Language Models","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Immune Cell Function and Interaction","field":"Immunology and Microbiology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Benchmark (surveying); Process (computing); Perspective (graphical); Measure (data warehouse); Focus (optics); Benchmarking; Service (business); User modeling","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01197048,0.002867212,0.001005292,0.005538287,0.001055768,0.002847003,0.00218828,0.002170933,0.002383939],"category_scores_gemma":[0.04549259,0.0004959154,0.00162109,0.003126475,0.0008637704,0.003792636,0.003308258,0.002060733,0.002322323],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001485509,"about_ca_system_score_gemma":0.001613471,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007973144,"about_ca_topic_score_gemma":0.01269608,"domain_scores_codex":[0.986482,0.006903541,0.001932795,0.001644321,0.002567987,0.000469304],"domain_scores_gemma":[0.969514,0.01862032,0.001635663,0.004862151,0.004166229,0.00120175],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003706824,0.006593511,0.1255738,0.005756089,0.002562468,0.001126014,0.002856715,0.1534868,0.02470155,0.01073117,0.1607027,0.5022024],"study_design_scores_gemma":[0.0007097922,0.00418463,0.05444304,0.0004565617,0.0004528658,0.001317701,0.001943892,0.840435,0.03251105,0.01334991,0.04980325,0.0003922865],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.6734629,0.004585638,0.1751165,0.001372287,0.0007065443,0.002835181,0.07516231,0.04678086,0.01997777],"genre_scores_gemma":[0.591023,0.0007174374,0.2171918,0.0006257379,0.0001464085,0.002424513,0.1831433,0.001585584,0.003142319],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01197048,"threshold_uncertainty_score":0.06330675,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03863659227809692,"score_gpt":0.3388043106272199,"score_spread":0.300167718349123,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}