{"id":"W4406892457","doi":"10.1109/fllm63129.2024.10852461","title":"Comparative Analysis of Loop-Free Function Evaluation Using ChatGPT and Copilot with C Bounded Model Checking","year":2024,"lang":"en","type":"article","venue":"","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université TÉLUQ","funders":"","keywords":"Bounded function; Loop (graph theory); Function (biology); Model checking; Computer science; Control theory (sociology); Mathematical optimization; Algorithm; Mathematics; Artificial intelligence; Mathematical analysis; Control (management); Combinatorics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000658842,0.000119526,0.0002591488,0.0002989417,0.0001159863,0.0001898009,0.0002282572,0.00003717357,0.00002297722],"category_scores_gemma":[0.00003421117,0.00009616553,0.0000443065,0.001464363,0.00006218596,0.0007454812,0.0001827203,0.0001542388,8.991641e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00008879857,"about_ca_system_score_gemma":0.0001406449,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001292953,"about_ca_topic_score_gemma":0.0001009948,"domain_scores_codex":[0.9987305,0.00009776088,0.0001936688,0.0003734886,0.0004744729,0.0001300405],"domain_scores_gemma":[0.9992487,0.0000980378,0.00009468447,0.0003475761,0.0001751894,0.00003578404],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001774901,0.00001131273,0.0005938206,0.00001660482,0.0004332484,8.246082e-7,0.001571913,0.9579358,0.0007093737,0.0348457,0.000007627985,0.003855997],"study_design_scores_gemma":[0.0002668761,0.00006277851,0.001231405,0.0000481782,0.0008213042,0.000002125003,0.0001020608,0.9930962,0.0002545797,0.003994836,0.000004103342,0.0001155411],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2102308,0.0001788364,0.7882581,0.00009124524,0.00007858168,0.0001232573,8.254036e-7,0.00008903901,0.0009492918],"genre_scores_gemma":[0.8751795,0.000002064705,0.124715,0.00002864583,0.00001769847,0.000004437917,0.00000468085,0.000005054153,0.00004287616],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6649487,"threshold_uncertainty_score":0.3921517,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06673743189225984,"score_gpt":0.3420002828752244,"score_spread":0.2752628509829645,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}