{"id":"W4392781618","doi":"10.48550/arxiv.2403.07059","title":"Better than classical? The subtle art of benchmarking quantum machine\\n learning models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Neural Networks and Applications","field":"Computer Science","cited_by":21,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Alliance de recherche numérique du Canada; University of Toronto; Innovation, Science and Economic Development Canada","keywords":"Benchmarking; Quantum; Computer science; Artificial intelligence; Cognitive science; Psychology; Physics; Economics; Quantum mechanics; Management","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008485705,0.0007479877,0.001323628,0.0008395943,0.001139845,0.003362182,0.003159288,0.002195434,0.005810427],"category_scores_gemma":[0.03499543,0.0004417714,0.0008534839,0.001177009,0.002782828,0.01047029,0.002383964,0.003713623,0.001370435],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001683241,"about_ca_system_score_gemma":0.001711339,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002969861,"about_ca_topic_score_gemma":0.004199168,"domain_scores_codex":[0.995282,0.002452602,0.0001979237,0.0006464513,0.001129427,0.0002916087],"domain_scores_gemma":[0.9895734,0.004666146,0.0003237743,0.004301448,0.0008126821,0.0003225262],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004980593,0.0004631057,0.006578316,0.00103425,0.0005540422,0.00009418627,0.0004682167,0.2852439,0.005225657,0.5761293,0.03548026,0.0882307],"study_design_scores_gemma":[0.00008142013,0.0002707231,0.001696157,0.0002164171,0.00006445582,0.00005534786,0.000242997,0.5471798,0.007282708,0.4201156,0.02270857,0.0000858714],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.4086283,0.008967553,0.4623812,0.03040389,0.002936878,0.0002455963,0.003195952,0.004892416,0.07834815],"genre_scores_gemma":[0.9082746,0.001862664,0.07922693,0.002896877,0.0003552337,0.0002863956,0.00218198,0.00125293,0.003662369],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008485705,"threshold_uncertainty_score":0.04487723,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08025898080473734,"score_gpt":0.1894861551547899,"score_spread":0.1092271743500526,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}