{"id":"W4415356660","doi":"10.1021/acs.jcim.5c01767","title":"dpdata: A Scalable Python Toolkit for Atomistic Machine Learning Data Sets","year":2025,"lang":"en","type":"article","venue":"Journal of Chemical Information and Modeling","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Centre for Innovation Studies","funders":"National Key Research and Development Program of China; National Science and Technology Major Project; Xiamen University; Agentúra na Podporu Výskumu a Vývoja; National Natural Science Foundation of China","keywords":"Python (programming language); Scalability; Inference; Software deployment; Data structure; Key (lock); Data point; Data set","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002601543,0.001827939,0.001513529,0.001706464,0.001412371,0.002797082,0.007049639,0.0008933384,0.03146821],"category_scores_gemma":[0.00986066,0.00153289,0.002227366,0.002225796,0.001191127,0.004460585,0.006457484,0.005016221,0.01901396],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001264052,"about_ca_system_score_gemma":0.00557401,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006862296,"about_ca_topic_score_gemma":0.00966143,"domain_scores_codex":[0.9983166,0.000281776,0.0001970739,0.0002655005,0.0007713424,0.0001677535],"domain_scores_gemma":[0.9974731,0.0007976241,0.0001623335,0.0007232132,0.0005757061,0.0002680513],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0008872545,0.0003520914,0.005243668,0.00236446,0.0006194466,0.0006517743,0.0007443632,0.04640843,0.01168455,0.05550626,0.712326,0.1632116],"study_design_scores_gemma":[0.0004626264,0.0001038223,0.002655977,0.0002572853,0.00008964724,0.0004285866,0.000185412,0.4253875,0.02572414,0.1320043,0.4123829,0.0003177495],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.004816327,0.0003060016,0.5389768,0.0006819587,0.0002902109,0.0004291503,0.05972366,0.3868891,0.0078866],"genre_scores_gemma":[0.06956776,0.0009702175,0.6535484,0.001386495,0.0001512369,0.004148781,0.1489157,0.1123745,0.008937061],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.03146821,"threshold_uncertainty_score":0.1052716,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03526463816445469,"score_gpt":0.3185870247305178,"score_spread":0.2833223865660631,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}