{"id":"W4413478130","doi":"10.36227/techrxiv.175606397.71672806/v1","title":"Evaluating Handwritten and Multimodal, Free-Style Responses in Algorithms and Data Structures: A RAG-LLM-Based Feedback Framework","year":2025,"lang":"en","type":"article","venue":"","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Science North","funders":"University of North Carolina at Charlotte","keywords":"Computer science; Style (visual arts); Algorithm; Speech recognition; Artificial intelligence; Natural language processing; Art","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008784484,0.002373471,0.0009609098,0.002649025,0.0004933862,0.002301443,0.001882597,0.001944309,0.006251751],"category_scores_gemma":[0.03465794,0.000367198,0.001165744,0.0006221685,0.0008054198,0.00265467,0.002914959,0.00161691,0.002841061],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001226116,"about_ca_system_score_gemma":0.001463934,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002521484,"about_ca_topic_score_gemma":0.005226858,"domain_scores_codex":[0.9921181,0.004097075,0.0004656122,0.001213755,0.001798919,0.000306636],"domain_scores_gemma":[0.9857244,0.007920316,0.001187771,0.001464439,0.002963789,0.0007393087],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002278054,0.001475001,0.01343139,0.001246977,0.0002511719,0.0003868285,0.002060872,0.0631764,0.07180957,0.004025794,0.01222822,0.8276298],"study_design_scores_gemma":[0.0002409628,0.003030154,0.01092374,0.0002023549,0.0001409094,0.0003213841,0.001057011,0.9066561,0.05484634,0.01109385,0.01127797,0.0002092593],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1656141,0.0009108989,0.7798525,0.000934815,0.0002440924,0.001589659,0.002027581,0.04264748,0.006178831],"genre_scores_gemma":[0.5921429,0.0001308305,0.3990395,0.0003747738,0.00008077693,0.0008823336,0.002064939,0.0007163367,0.004567659],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.008784484,"threshold_uncertainty_score":0.04645735,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05574757992907353,"score_gpt":0.3856549634586867,"score_spread":0.3299073835296131,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}