{"id":"W7082255410","doi":"10.48448/v1pp-6e61","title":"ChatBench: From Static Benchmarks to Human-AI Evaluation","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Measure (data warehouse); Conversation; Isolation (microbiology); Scale (ratio); Carry (investment); Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01595811,0.003942022,0.001516659,0.006778311,0.001557443,0.004222796,0.003733052,0.002884631,0.005265611],"category_scores_gemma":[0.08654404,0.0007989111,0.000964458,0.003699752,0.001711458,0.00577507,0.005816131,0.002956087,0.008118041],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002346346,"about_ca_system_score_gemma":0.001792428,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01010795,"about_ca_topic_score_gemma":0.01325438,"domain_scores_codex":[0.9710445,0.01651245,0.001965917,0.00428957,0.005165752,0.001021751],"domain_scores_gemma":[0.9258872,0.04191115,0.003690033,0.01438823,0.009816514,0.004306986],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.005831887,0.005019358,0.09443882,0.006083945,0.001308928,0.0005866979,0.004690786,0.04909177,0.0135554,0.01123955,0.4419058,0.366247],"study_design_scores_gemma":[0.001129379,0.003867065,0.09197468,0.001213479,0.0002872761,0.0008317646,0.004958139,0.667684,0.02585833,0.03803874,0.1635629,0.0005943525],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5102072,0.01319328,0.1530231,0.004553889,0.002888097,0.003445307,0.1165049,0.1459465,0.05023754],"genre_scores_gemma":[0.6912877,0.0008712095,0.1138353,0.001831948,0.0005848438,0.003617312,0.1740395,0.00573026,0.008201872],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01595811,"threshold_uncertainty_score":0.08439559,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02535826857538702,"score_gpt":0.3210684137118414,"score_spread":0.2957101451364543,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}