{"id":"W2108653655","doi":"10.1109/fie.2006.322402","title":"A Tool for Automated GUI Program Grading","year":2006,"lang":"en","type":"article","venue":"","topic":"Software Engineering Research","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of New Brunswick","funders":"","keywords":"Computer science; Java; Graphical user interface; Graphical user interface testing; Software engineering; Grading (engineering); Programming language; Consistency (knowledge bases); Flexibility (engineering); User interface; Operating system; User interface design; Artificial intelligence; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00692102,0.002784796,0.001780891,0.006428197,0.000966525,0.003725815,0.003933582,0.001834539,0.02895561],"category_scores_gemma":[0.03672753,0.001965992,0.001586793,0.002546898,0.0006473502,0.005157504,0.003674502,0.00308633,0.02291958],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009651451,"about_ca_system_score_gemma":0.001520346,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001449595,"about_ca_topic_score_gemma":0.001499268,"domain_scores_codex":[0.9918133,0.001806712,0.001317582,0.001360618,0.003309919,0.0003919023],"domain_scores_gemma":[0.9778676,0.009572729,0.001439066,0.00548045,0.004980621,0.0006594694],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0005785626,0.0004623118,0.00294912,0.0006133667,0.0001101252,0.0004618742,0.0004963227,0.004389636,0.01295771,0.008758762,0.1251697,0.8430526],"study_design_scores_gemma":[0.001181978,0.0008082495,0.008618664,0.0009642413,0.0002317507,0.003900406,0.0003153105,0.2817256,0.125384,0.05079987,0.5253187,0.0007512713],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00255365,0.0001298545,0.6442345,0.0001360097,0.0001287013,0.0003508395,0.001849805,0.3469638,0.00365285],"genre_scores_gemma":[0.04963462,0.0002417429,0.8834417,0.0003193253,0.0001147233,0.001154089,0.01280829,0.0408184,0.01146708],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02895561,"threshold_uncertainty_score":0.09686619,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01592801537558914,"score_gpt":0.300303994272294,"score_spread":0.2843759788967049,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}