{"id":"W1785571949","doi":"10.22329/celt.v8i0.4244","title":"Integrated Testlets: A New Form of Expert-Student Collaborative Testing","year":2015,"lang":"en","type":"article","venue":"Collected Essays on Learning and Teaching","topic":"Educational Assessment and Pedagogy","field":"Social Sciences","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"Trent University","funders":"Brock University; Trent University","keywords":"Formative assessment; Set (abstract data type); Computer science; Test (biology); Multiple choice; Psychology; Mathematics education","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01333267,0.001086684,0.001044286,0.003388498,0.0006557476,0.003106252,0.002972551,0.001404182,0.01081353],"category_scores_gemma":[0.05443774,0.0006266145,0.0006615713,0.001973642,0.001025918,0.00374266,0.004997267,0.001958069,0.00369721],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006966203,"about_ca_system_score_gemma":0.001718729,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000611588,"about_ca_topic_score_gemma":0.001732484,"domain_scores_codex":[0.9781345,0.01082458,0.00157898,0.002168889,0.006698472,0.0005946509],"domain_scores_gemma":[0.9339841,0.03759222,0.003232044,0.01307243,0.009632468,0.002486811],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.0007017407,0.003284435,0.01306001,0.0004631357,0.00008736239,0.0005276941,0.004629084,0.001787162,0.01766416,0.006985011,0.01860362,0.9322066],"study_design_scores_gemma":[0.001514558,0.015784,0.1430548,0.001636325,0.0003744916,0.01288008,0.005002448,0.08555273,0.09010728,0.0799479,0.5630088,0.001136624],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1133811,0.0003697558,0.8255403,0.001197938,0.0007737668,0.005789733,0.002206162,0.01173002,0.03901123],"genre_scores_gemma":[0.1836688,0.0002328403,0.7923113,0.000536471,0.0002689534,0.004215766,0.002064836,0.0008480784,0.01585293],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01333267,"threshold_uncertainty_score":0.07051069,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08347668774675397,"score_gpt":0.4111060133292314,"score_spread":0.3276293255824774,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}