{"id":"W4382515967","doi":"10.1145/3587103.3594211","title":"Conducting Multi-Institutional Studies of Parsons Problems","year":2023,"lang":"en","type":"article","venue":"","topic":"Teaching and Learning Programming","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Scratch; Computer science; Code (set theory); Coding (social sciences); Publication; Order (exchange); Programming language; Mathematics education; Software engineering; Human–computer interaction; Psychology; Sociology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02513321,0.000727688,0.001134868,0.006537549,0.004447472,0.00457682,0.002558097,0.001799944,0.004859921],"category_scores_gemma":[0.05907857,0.0007724606,0.001320162,0.006641141,0.002506502,0.007844465,0.005223324,0.003544948,0.001405222],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004869815,"about_ca_system_score_gemma":0.009004237,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003521168,"about_ca_topic_score_gemma":0.01516903,"domain_scores_codex":[0.9737008,0.01437327,0.002995582,0.00230595,0.004495376,0.002129098],"domain_scores_gemma":[0.8885785,0.05086459,0.01812643,0.009974077,0.02757747,0.004878869],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.00171797,0.02603647,0.31691,0.01020403,0.001605611,0.001941741,0.1637979,0.001998168,0.008016986,0.0125864,0.02590569,0.4292788],"study_design_scores_gemma":[0.0008393913,0.01061969,0.3889296,0.006445815,0.001258853,0.001636172,0.4238306,0.001582309,0.02452705,0.01056868,0.1292862,0.000475633],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9638157,0.004203353,0.007464571,0.002783864,0.0002637792,0.00379659,0.0007655494,0.0001728485,0.0167336],"genre_scores_gemma":[0.9614586,0.004531817,0.01614933,0.002472137,0.0001888198,0.008908928,0.0009259713,0.0001016127,0.005262854],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02513321,"threshold_uncertainty_score":0.1329187,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3230945578174528,"score_gpt":0.3792800844996193,"score_spread":0.05618552668216648,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}