{"id":"W2094237702","doi":"10.1119/1.4820241","title":"Integrated testlets and the immediate feedback assessment technique","year":2013,"lang":"en","type":"article","venue":"American Journal of Physics","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":22,"is_retracted":false,"has_abstract":true,"ca_institutions":"Trent University","funders":"","keywords":"Context (archaeology); Multiple choice; Set (abstract data type); Test (biology); Reliability (semiconductor); Standard deviation; Computer science; Mathematics education; Statistics; Physics; Psychology; Mathematics; Programming language; Significant difference","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009211964,0.0009524939,0.0007129935,0.002094987,0.0003500202,0.002119419,0.002263404,0.001267733,0.01154826],"category_scores_gemma":[0.05758725,0.0006038671,0.0007265042,0.001506816,0.0008384455,0.002364619,0.003236331,0.002208028,0.004022092],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00056701,"about_ca_system_score_gemma":0.00140837,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004705482,"about_ca_topic_score_gemma":0.0005435972,"domain_scores_codex":[0.9835616,0.007639069,0.0008500791,0.001736076,0.005671698,0.0005415077],"domain_scores_gemma":[0.9543226,0.02795124,0.003138593,0.006616712,0.006738453,0.001232421],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.001261977,0.001128789,0.008967131,0.0005930545,0.0000766598,0.0004101701,0.001673871,0.002798203,0.01674055,0.01590895,0.008758062,0.9416826],"study_design_scores_gemma":[0.001645304,0.02921167,0.1666264,0.002815615,0.0005981146,0.01885192,0.001900834,0.1390827,0.1131875,0.1563474,0.368839,0.0008935151],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07620105,0.0005327561,0.8907543,0.0006701187,0.0003554762,0.002267338,0.0006549741,0.004694826,0.02386919],"genre_scores_gemma":[0.2302466,0.0002710546,0.7562872,0.0005773728,0.0002807662,0.002294377,0.001071922,0.000393945,0.008576756],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01154826,"threshold_uncertainty_score":0.04871809,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01254507973410524,"score_gpt":0.314171238454659,"score_spread":0.3016261587205538,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}