{"id":"W6894293496","doi":"10.5683/sp3/qlyaco","title":"Benchmark synthesis","year":2024,"lang":"en","type":"dataset","venue":"Borealis","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Sherbrooke","funders":"","keywords":"Inference; Benchmark (surveying); Artificial neural network; Latency (audio); Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007033511,0.003434626,0.0009790484,0.00228623,0.0006754533,0.00119669,0.003104508,0.001659422,0.0238969],"category_scores_gemma":[0.003664774,0.0004924266,0.00136714,0.002897041,0.0004420668,0.0009300635,0.0009775087,0.001149477,0.01561828],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001597386,"about_ca_system_score_gemma":0.001816404,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02073538,"about_ca_topic_score_gemma":0.04863724,"domain_scores_codex":[0.9991239,0.000120929,0.00008778841,0.0002713782,0.0002751777,0.0001209387],"domain_scores_gemma":[0.998882,0.0002663115,0.00006739623,0.0003165201,0.0003964849,0.00007126314],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0007977262,0.0004325122,0.002694017,0.002748627,0.0003034785,0.0004021929,0.00004326708,0.0444186,0.004151663,0.002178105,0.8520075,0.08982236],"study_design_scores_gemma":[0.001273212,0.0009424794,0.0106012,0.0005973235,0.0003352735,0.001198038,0.0002697649,0.1172377,0.0255673,0.01176996,0.8300462,0.0001615085],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.04688899,0.004410367,0.01012169,0.0008146405,0.0009775592,0.0005081484,0.8735532,0.01929917,0.04342629],"genre_scores_gemma":[0.03354958,0.000690334,0.01060686,0.0002369362,0.0000481109,0.0003447612,0.9436311,0.0006036223,0.01028864],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.0238969,"threshold_uncertainty_score":0.07994312,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01644723638811268,"score_gpt":0.2714038769552922,"score_spread":0.2549566405671795,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}