{"id":"W4416767877","doi":"10.5539/jel.v15n2p91","title":"Multidimensional Diagnostic Technique for Mathematical Proficiency with Automated Feedback Generation","year":2025,"lang":"","type":"article","venue":"Journal of Education and Learning","topic":"Mathematics Education and Teaching Techniques","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Research Council of Thailand","keywords":"Formative assessment; Item response theory; Reliability (semiconductor); Novelty; Construct (python library); Computerized adaptive testing; Binary number; Scalability","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01373934,0.001078237,0.0006121364,0.004852644,0.0004029986,0.001663707,0.001162463,0.0007718061,0.004649715],"category_scores_gemma":[0.07050308,0.0003006702,0.0006778454,0.001987465,0.0007079719,0.001977759,0.002610244,0.0009475008,0.00173184],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006355543,"about_ca_system_score_gemma":0.0008575043,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000746647,"about_ca_topic_score_gemma":0.00102428,"domain_scores_codex":[0.9804781,0.01018909,0.001885509,0.001788083,0.005288376,0.0003708757],"domain_scores_gemma":[0.9385083,0.03567128,0.006338079,0.00699295,0.01197145,0.0005178954],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001170963,0.0005156343,0.04829994,0.0007640152,0.0001391883,0.0002388383,0.006441244,0.004828117,0.04484915,0.009511126,0.004146003,0.8790958],"study_design_scores_gemma":[0.0005234259,0.004216516,0.2105731,0.001184834,0.0003128944,0.00364677,0.009754684,0.4816051,0.1966449,0.0376115,0.05294671,0.0009794619],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.169811,0.0001582274,0.8158788,0.0002970464,0.0001487309,0.001617617,0.0007957109,0.005731676,0.00556131],"genre_scores_gemma":[0.4335132,0.00009143034,0.5626123,0.0001275816,0.00003836652,0.001662535,0.0004888213,0.0002222786,0.001243533],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01373934,"threshold_uncertainty_score":0.07266146,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02255915278780524,"score_gpt":0.3690986576693623,"score_spread":0.3465395048815571,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}