{"id":"W4405935413","doi":"10.23919/cnsm62983.2024.10814616","title":"Evaluating the Robustness of ADVENT on the VeReMi-Extension Dataset","year":2024,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"New York Institute of Technology; Dalhousie University","funders":"","keywords":"Robustness (evolution); Extension (predicate logic); Computer science; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001115106,0.00007265026,0.00006134024,0.00003406891,0.00009556401,0.0001386777,0.001037527,0.00002493727,0.00003309904],"category_scores_gemma":[0.0001756441,0.00002879939,0.0000300463,0.0003084634,0.00003954325,0.0001711254,0.0003775833,0.000177575,0.00001360423],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00001847636,"about_ca_system_score_gemma":0.00004610365,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00002226553,"about_ca_topic_score_gemma":0.000002263155,"domain_scores_codex":[0.9991139,0.00008969026,0.0001270924,0.0002193926,0.0003450553,0.0001048517],"domain_scores_gemma":[0.9987472,0.0004341155,0.00003719264,0.0007169145,0.00005145003,0.00001315095],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001121449,0.0000419612,0.000001808931,0.0000904854,0.00002152614,0.0000160584,0.0004540691,0.001097417,0.01725483,0.4931082,0.1154883,0.3724141],"study_design_scores_gemma":[0.00003533256,0.0001372389,0.00001650159,0.0002949494,0.0000103912,0.00002157511,0.0000343773,0.9172263,0.0621558,0.01770244,0.00226518,0.00009988595],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01831603,0.005501633,0.9483124,0.02465479,0.0008241716,0.0005399588,0.00004980219,0.0009364024,0.0008647991],"genre_scores_gemma":[0.8112878,0.00001257362,0.1871915,0.001157023,0.00008008957,0.00001867224,0.0000238024,0.000009108737,0.0002194387],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9161289,"threshold_uncertainty_score":0.1928,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0751668046710298,"score_gpt":0.3836627307274215,"score_spread":0.3084959260563917,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}