{"id":"W4412512602","doi":"10.26434/chemrxiv-2025-vj4fk","title":"How the Discreteness of the Periodic Table Undermines Machine Learning Predictions","year":2025,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Analytical Chemistry and Chromatography","field":"Chemistry","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto; Government of Ontario; Compute Canada","keywords":"Table (database); Computer science; Artificial intelligence; Periodic table; Machine learning; Statistical physics; Cognitive science; Psychology; Physics; Data mining; Quantum mechanics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004105297,0.0006651401,0.001042505,0.001077207,0.0007508848,0.00365565,0.001981347,0.001568942,0.002404359],"category_scores_gemma":[0.02859167,0.0007498965,0.0008661176,0.0007732051,0.002479116,0.005169575,0.001581618,0.002958352,0.0008787069],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001106334,"about_ca_system_score_gemma":0.001168248,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005089266,"about_ca_topic_score_gemma":0.00462635,"domain_scores_codex":[0.9983903,0.0005451959,0.00008876665,0.00055447,0.0003148345,0.0001064783],"domain_scores_gemma":[0.9865741,0.009187088,0.0007368096,0.002682885,0.0004853807,0.0003337495],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008745738,0.0002638265,0.05441415,0.0008662133,0.0004029167,0.0004922338,0.0005271061,0.7155526,0.01117066,0.08637918,0.01990369,0.1091529],"study_design_scores_gemma":[0.00003420342,0.00005879515,0.002830199,0.00006689849,0.00002698246,0.00009262875,0.00008032705,0.8688757,0.002629756,0.1229758,0.002301289,0.00002730752],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6848032,0.003695162,0.2735433,0.01377788,0.000605878,0.00008736541,0.005812874,0.003982918,0.01369141],"genre_scores_gemma":[0.9650978,0.0006741924,0.02923148,0.0008676755,0.0001249489,0.00005044484,0.002820038,0.0002380793,0.0008953384],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005089266,"threshold_uncertainty_score":0.02171117,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0143172629765192,"score_gpt":0.2265692293375964,"score_spread":0.2122519663610772,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}