{"id":"W2895868736","doi":"10.3233/aic-170537","title":"Performance robustness of AI planners in the 2014 International Planning Competition","year":2018,"lang":"en","type":"article","venue":"AI Communications","topic":"Artificial Intelligence in Games","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada","keywords":"Agile software development; Solver; Computer science; Robustness (evolution); Software; Benchmark (surveying); Competition (biology); Homogeneous; Software engineering; Operating system","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007297532,0.001197313,0.0009855654,0.001729228,0.001131337,0.002583896,0.001614264,0.001405246,0.004526457],"category_scores_gemma":[0.02548354,0.000565248,0.001025413,0.001914252,0.001653791,0.002060873,0.002369245,0.002167651,0.000926083],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003461957,"about_ca_system_score_gemma":0.004767125,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02604051,"about_ca_topic_score_gemma":0.03099045,"domain_scores_codex":[0.995154,0.00158079,0.0002799375,0.0009595832,0.001118772,0.0009068521],"domain_scores_gemma":[0.9888697,0.006645857,0.0006082974,0.001208892,0.001370682,0.001296527],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002528703,0.000731581,0.01949255,0.0003905089,0.0004114629,0.0001509444,0.0002929774,0.86488,0.001633838,0.0113044,0.02973451,0.06844842],"study_design_scores_gemma":[0.0007575862,0.001801134,0.02435158,0.0001402578,0.0001546203,0.0001636648,0.001127056,0.928846,0.004635922,0.0184321,0.01948673,0.0001033778],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9308769,0.001552115,0.01087743,0.00188288,0.0004540254,0.0001701148,0.003965874,0.002022245,0.04819842],"genre_scores_gemma":[0.9736465,0.0003312524,0.01281846,0.0003028189,0.00005574154,0.0001318129,0.009124649,0.0003265186,0.003262345],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02604051,"threshold_uncertainty_score":0.0517779,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06033374970487245,"score_gpt":0.3527300751938626,"score_spread":0.2923963254889901,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}