{"id":"W4401282765","doi":"10.1007/978-3-031-66997-2_3","title":"Using General Large Language Models to Classify Mathematical Documents","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Mathematics, Computing, and Information Processing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Natural language processing; Artificial intelligence; Information retrieval; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002723394,0.0009359033,0.0009438777,0.003926865,0.001111142,0.004514095,0.001588965,0.001339226,0.003310934],"category_scores_gemma":[0.009263982,0.0004653539,0.001978875,0.00235637,0.001177596,0.007419915,0.001814451,0.002529648,0.002673246],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001496906,"about_ca_system_score_gemma":0.001394369,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00249644,"about_ca_topic_score_gemma":0.003649302,"domain_scores_codex":[0.9978592,0.0007954613,0.0002415447,0.0004566486,0.0004969356,0.0001500646],"domain_scores_gemma":[0.992382,0.005035364,0.0005607735,0.001054291,0.0007278775,0.000239698],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008530268,0.0007270804,0.01223673,0.0008249302,0.000362756,0.0005390095,0.001219684,0.06069146,0.01630085,0.1867285,0.03466694,0.6848491],"study_design_scores_gemma":[0.00004872544,0.0001358532,0.001480615,0.00009814839,0.0001190534,0.0003160656,0.0002760015,0.6789276,0.004199558,0.3033243,0.0110183,0.00005581542],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06254454,0.002422385,0.9202989,0.001802666,0.0002148973,0.0002231021,0.003016696,0.003869499,0.005607408],"genre_scores_gemma":[0.5939254,0.001584886,0.376515,0.000855075,0.0007029322,0.00049212,0.01306926,0.0006613357,0.01219408],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004514095,"threshold_uncertainty_score":0.01440287,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03733455946854739,"score_gpt":0.302462580135229,"score_spread":0.2651280206666816,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}