{"id":"W7129036557","doi":"10.1109/icet67421.2025.11380681","title":"A Mathematics Question Diagram Generation Strategy Defined with Codes and Empowered by Reasoning Large Language Models","year":2025,"lang":"","type":"article","venue":"","topic":"Mathematics, Computing, and Information Processing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Diagram; Function (biology); Code (set theory); Quality (philosophy); Code generation; Encoding (memory)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002723017,0.001264386,0.0005212739,0.00214915,0.0006442189,0.002366277,0.001647518,0.00109879,0.01313031],"category_scores_gemma":[0.0149894,0.0005653164,0.001058789,0.0009475793,0.001180907,0.003209098,0.003329939,0.001654253,0.004716531],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008466673,"about_ca_system_score_gemma":0.00252567,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002227541,"about_ca_topic_score_gemma":0.002964981,"domain_scores_codex":[0.9972581,0.001186361,0.0001940374,0.0006396199,0.0006288532,0.00009303926],"domain_scores_gemma":[0.9937862,0.00305126,0.0002568838,0.00131439,0.001313258,0.0002780998],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004674617,0.0005828945,0.003578705,0.001184667,0.00009230174,0.0008162591,0.008327626,0.01605065,0.04167892,0.2327446,0.04083283,0.653643],"study_design_scores_gemma":[0.0002355691,0.0004087461,0.001161994,0.0003764036,0.00009808366,0.001360846,0.001950576,0.492456,0.07662676,0.1504338,0.2746942,0.0001970286],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.004858125,0.00002408698,0.9790973,0.000203731,0.00003149201,0.0003535332,0.000407863,0.01140482,0.003619103],"genre_scores_gemma":[0.05341088,0.00003472515,0.9373065,0.0001311025,0.00001231713,0.0005208864,0.001271198,0.001778624,0.005533806],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01313031,"threshold_uncertainty_score":0.04392523,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01339690474757257,"score_gpt":0.2659466993643034,"score_spread":0.2525497946167309,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}