{"id":"W1981334977","doi":"10.1007/s10472-010-9191-0","title":"Tradeoffs in the empirical evaluation of competing algorithm designs","year":2010,"lang":"en","type":"article","venue":"Annals of Mathematics and Artificial Intelligence","topic":"Constraint Satisfaction and Optimization","field":"Computer Science","cited_by":37,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Algorithm; Parameterized complexity; Set (abstract data type); Algorithm design; Integer programming","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2826677,0.002684927,0.004322752,0.00523378,0.001676689,0.00743626,0.005607565,0.00844013,0.00410611],"category_scores_gemma":[0.6881793,0.002706295,0.003018905,0.004303687,0.006681432,0.01228031,0.005691728,0.006778096,0.0005169469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004995259,"about_ca_system_score_gemma":0.004075077,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001652335,"about_ca_topic_score_gemma":0.001806923,"domain_scores_codex":[0.6250473,0.3425776,0.009535678,0.005246369,0.01632258,0.001270458],"domain_scores_gemma":[0.1128626,0.8660169,0.004974918,0.0102364,0.004766482,0.001142701],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006226404,0.0008949479,0.01886803,0.001865151,0.002229953,0.000215005,0.001353239,0.5641229,0.000848607,0.2096616,0.003553433,0.1901607],"study_design_scores_gemma":[0.0009861911,0.001845826,0.002603229,0.0004261743,0.0004801801,0.0002729026,0.000220817,0.8165006,0.0008091072,0.1739394,0.001790841,0.0001247423],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1442711,0.01086534,0.8301297,0.00461931,0.0003102182,0.000806206,0.0003156584,0.0004200412,0.00826235],"genre_scores_gemma":[0.6218612,0.001162824,0.3729065,0.000664536,0.0002232567,0.001079968,0.000457306,0.0003308321,0.001313556],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.2826677,"threshold_uncertainty_score":0.8845985,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4438438458895254,"score_gpt":0.4265607545525879,"score_spread":0.01728309133693751,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}