{"id":"W4410390504","doi":"10.1007/s41870-025-02555-4","title":"Assessing GPT model uncertainty in mathematical OCR tasks via entropy analysis","year":2025,"lang":"en","type":"article","venue":"International Journal of Information Technology","topic":"Reservoir Engineering and Simulation Methods","field":"Engineering","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Wilfrid Laurier University","funders":"","keywords":"Computer science; Entropy (arrow of time); Artificial intelligence; Data mining; Machine learning","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003961962,0.0008083524,0.0007421379,0.001176522,0.0004830217,0.001636682,0.0008569987,0.001324173,0.002233703],"category_scores_gemma":[0.03312875,0.0003470569,0.0007646045,0.0006876523,0.00096529,0.002964304,0.002025239,0.001416956,0.0002470507],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001160142,"about_ca_system_score_gemma":0.001274731,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006636462,"about_ca_topic_score_gemma":0.003163279,"domain_scores_codex":[0.9984534,0.0007342965,0.00008855427,0.0002047767,0.000389989,0.0001289268],"domain_scores_gemma":[0.9699921,0.02700432,0.001026585,0.0009161004,0.0007940658,0.0002668377],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001443115,0.00004235328,0.001508484,0.00006448626,0.00003965619,0.0000503031,0.00006162406,0.9657124,0.001183732,0.004943803,0.0002933187,0.02595551],"study_design_scores_gemma":[0.000002974155,0.00002350121,0.0002922809,0.000004389911,0.000004136667,0.000009104145,0.000009597306,0.9952832,0.0004729776,0.003849154,0.00004345785,0.000005185635],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3413204,0.0002897073,0.6515342,0.0005883751,0.00004286202,0.00009594639,0.00032355,0.0008005103,0.005004421],"genre_scores_gemma":[0.965688,0.00006557342,0.03320849,0.00005805989,0.00001712377,0.00005953016,0.0002363544,0.00008246082,0.0005844273],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.006636462,"threshold_uncertainty_score":0.02095312,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.009238933779717358,"score_gpt":0.3115749087716572,"score_spread":0.3023359749919398,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}