{"id":"W6969185504","doi":"10.5683/sp3/kus7ob","title":"Replication Data for: Integrating C-H Information to Improve Machine Learning Classification Models for Microplastic Identification from Raman Spectra","year":2024,"lang":"en","type":"dataset","venue":"Borealis","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Raman spectroscopy; Python (programming language); Identification (biology); Noise (video); Scaling; Pattern recognition (psychology); Missing data","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002522576,0.002788384,0.001367631,0.002668008,0.001498324,0.002171861,0.004943548,0.002475446,0.0223104],"category_scores_gemma":[0.00841534,0.0006514144,0.002186806,0.002788604,0.0008733523,0.001644893,0.002741917,0.002336805,0.05000854],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001566728,"about_ca_system_score_gemma":0.002343128,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02928004,"about_ca_topic_score_gemma":0.06563786,"domain_scores_codex":[0.9977294,0.0003611126,0.0001986565,0.0007237545,0.0007006147,0.0002864807],"domain_scores_gemma":[0.996367,0.0005216801,0.0001747081,0.001567715,0.001122317,0.0002465587],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003526125,0.0002049498,0.005500923,0.0007797747,0.000113201,0.0001336326,0.00007901761,0.001775965,0.001316607,0.0007285892,0.9713821,0.01763263],"study_design_scores_gemma":[0.0004943463,0.00013295,0.01622498,0.0003085693,0.0001019319,0.000327705,0.0003378575,0.006630011,0.005214435,0.004023108,0.9660559,0.0001482764],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.004329486,0.0003523377,0.001899257,0.0003361931,0.0003255161,0.0001385357,0.9843796,0.004727974,0.003511025],"genre_scores_gemma":[0.00405848,0.00006363915,0.003943838,0.00009648241,0.00002362807,0.0001581902,0.9900272,0.0002274006,0.001400962],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.02928004,"threshold_uncertainty_score":0.07463574,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0498944940288475,"score_gpt":0.3195180276281993,"score_spread":0.2696235335993518,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}