{"id":"W4390940921","doi":"10.1038/s41586-023-06747-5","title":"Solving olympiad geometry without human demonstrations","year":2024,"lang":"en","type":"article","venue":"Nature","topic":"Logic, programming, and type systems","field":"Computer Science","cited_by":254,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University","keywords":"Olympiad; Mathematical proof; Automated theorem proving; Computer science; Gas meter prover; Euclidean geometry; Artificial intelligence; Calculus (dental); Mathematics; Theoretical computer science; Geometry; Mathematics education","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001703415,0.001174545,0.0007049959,0.0006870319,0.0005954624,0.001429266,0.002130432,0.001351163,0.01133962],"category_scores_gemma":[0.01575596,0.0004894863,0.001024997,0.00046779,0.001890687,0.002444976,0.002962956,0.001650946,0.002020749],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00131434,"about_ca_system_score_gemma":0.002258821,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003277735,"about_ca_topic_score_gemma":0.005493274,"domain_scores_codex":[0.9980432,0.0006866442,0.0001233357,0.0004636244,0.0005585104,0.0001246957],"domain_scores_gemma":[0.9904531,0.00639834,0.0004206846,0.001685794,0.0007930337,0.0002489492],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008626052,0.0004891934,0.009103036,0.002428961,0.0002554153,0.0009161046,0.000897873,0.3201382,0.02328742,0.05085807,0.05024327,0.5405199],"study_design_scores_gemma":[0.000228883,0.0003222937,0.002392999,0.0002223286,0.0000611041,0.0005121125,0.0005046246,0.8621164,0.02944597,0.0568955,0.04724842,0.00004944923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3375673,0.002097625,0.58489,0.003309725,0.0005728754,0.0004667934,0.003738458,0.02705913,0.04029805],"genre_scores_gemma":[0.6170235,0.0004238971,0.3740482,0.0003892025,0.00005824426,0.0001300295,0.003248778,0.0007360977,0.003941917],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01133962,"threshold_uncertainty_score":0.03793484,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01459979506797145,"score_gpt":0.2838805558926104,"score_spread":0.2692807608246389,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}