{"id":"W6929445656","doi":"10.48448/me00-k761","title":"A User-Centric Multi-Intent Benchmark for Evaluating Large Language Models","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Immune Cell Function and Interaction","field":"Immunology and Microbiology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Benchmark (surveying); Process (computing); Perspective (graphical); Measure (data warehouse); Focus (optics); Benchmarking; Service (business); User modeling","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0008427882,0.0002840563,0.0003105677,0.0007390783,0.0002191039,0.00008434293,0.0004135265,0.0003259627,0.005024555],"category_scores_gemma":[0.0002282003,0.0002367584,0.0001442698,0.0003994296,0.0002408784,0.0001534237,0.0001852865,0.0003497333,0.002818699],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001833462,"about_ca_system_score_gemma":0.0002433149,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002171148,"about_ca_topic_score_gemma":0.0001295063,"domain_scores_codex":[0.9982119,0.00006521476,0.0003373721,0.0006810417,0.0001126642,0.0005918694],"domain_scores_gemma":[0.9991429,0.00006184545,0.0002096364,0.0003947592,0.0001575621,0.00003323835],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00005869882,0.0002927425,0.000006217782,0.0001492996,0.0001745016,0.000003773735,0.0007943883,0.00004116659,0.08851203,0.003885731,0.8989412,0.007140314],"study_design_scores_gemma":[0.001690801,0.0002820068,0.000004218395,0.0003773327,0.0001577868,0.00003928431,0.001885685,0.03779808,0.004643897,0.0001391531,0.9525266,0.0004551255],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.001147724,0.06635414,0.3079187,0.0008881086,0.08272901,0.005449617,0.0006415427,0.00228694,0.5325842],"genre_scores_gemma":[0.01617698,0.00009726676,0.002599613,0.0003730235,0.0001429797,0.00007702366,0.0005145367,0.0001918162,0.9798267],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.4472425,"threshold_uncertainty_score":0.9979577,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03863659227809692,"score_gpt":0.3388043106272199,"score_spread":0.300167718349123,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}