{"id":"W4405935413","doi":"10.23919/cnsm62983.2024.10814616","title":"Evaluating the Robustness of ADVENT on the VeReMi-Extension Dataset","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"New York Institute of Technology; Dalhousie University","funders":"","keywords":"Robustness (evolution); Extension (predicate logic); Computer science; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006058435,0.002248355,0.001184283,0.004323684,0.0009724703,0.002050746,0.002048021,0.00181006,0.0009526972],"category_scores_gemma":[0.0105127,0.0002554943,0.001630303,0.001699593,0.0008107639,0.002809209,0.001626221,0.001331437,0.001140782],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001114494,"about_ca_system_score_gemma":0.001065608,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01584315,"about_ca_topic_score_gemma":0.02612849,"domain_scores_codex":[0.9952441,0.001021078,0.000555507,0.001483875,0.001375144,0.0003203368],"domain_scores_gemma":[0.9955518,0.001980118,0.0003565557,0.001103377,0.0007708276,0.0002373978],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003660632,0.003552475,0.1300284,0.002591201,0.00262784,0.001044166,0.0006048789,0.2545565,0.03073759,0.00420591,0.1824384,0.3839519],"study_design_scores_gemma":[0.0004267042,0.001718292,0.07859764,0.0002769163,0.0004059992,0.001777551,0.0009934738,0.8028241,0.02922257,0.003786833,0.07969905,0.0002708261],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8310583,0.005604504,0.02667143,0.001595721,0.00142658,0.001082321,0.1026105,0.01550743,0.01444322],"genre_scores_gemma":[0.5402078,0.0008568711,0.05856302,0.0005471734,0.0002925816,0.0003135367,0.393003,0.0004919869,0.005724038],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01584315,"threshold_uncertainty_score":0.03204048,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0751668046710298,"score_gpt":0.3836627307274215,"score_spread":0.3084959260563917,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}