{"id":"W4221057764","doi":"10.1190/int-2021-0194.1","title":"Generating a labeled data set to train machine learning algorithms for lithologic classification of drill cuttings","year":2022,"lang":"en","type":"article","venue":"Interpretation","topic":"Mineral Processing and Grinding","field":"Engineering","cited_by":11,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Workflow; Convolutional neural network; Lithology; Artificial intelligence; Algorithm; Geology; Data set; Computer science; Set (abstract data type); Drill; Machine learning; Pattern recognition (psychology); Database; Petrology; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001822526,0.001719541,0.0006653867,0.001935718,0.0008848588,0.001277499,0.002245688,0.002092785,0.003637106],"category_scores_gemma":[0.007375614,0.0005604564,0.001312954,0.001410757,0.0008753059,0.001413837,0.001172079,0.001654463,0.003074579],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001912576,"about_ca_system_score_gemma":0.001920823,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01809184,"about_ca_topic_score_gemma":0.02208936,"domain_scores_codex":[0.9986432,0.0002763602,0.0001069631,0.0004743914,0.0003295668,0.0001695729],"domain_scores_gemma":[0.9946598,0.001726943,0.0003861745,0.001035844,0.002032845,0.0001583352],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0008736552,0.002177987,0.04377721,0.0006887399,0.0003407225,0.0007280908,0.0002896214,0.3693007,0.03393707,0.004984056,0.05495751,0.4879445],"study_design_scores_gemma":[0.00005607886,0.0001622542,0.005488823,0.00005302922,0.00003109013,0.00006799924,0.0001707627,0.95838,0.02583996,0.003674595,0.006039758,0.00003557892],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.5485374,0.0007021578,0.3892058,0.0009822607,0.0004885948,0.0013491,0.02692554,0.02163963,0.01016946],"genre_scores_gemma":[0.5028548,0.000177631,0.4247802,0.0004162973,0.00008320978,0.001347241,0.06606856,0.0004515669,0.003820515],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.01809184,"threshold_uncertainty_score":0.03597307,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06237746096804688,"score_gpt":0.3144729504777315,"score_spread":0.2520954895096847,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}