{"id":"W4389095557","doi":"10.1007/978-3-031-49252-5_6","title":"IDPP: Imbalanced Datasets Pipelines in Pyrus","year":2023,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Computer science; Python (programming language); Pipeline transport; Pipeline (software); Preprocessor; Code reuse; Code (set theory); Data mining; Analytics; Machine learning; Programming language; Database; Software engineering; Software","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.01004538,0.0005095815,0.0007470968,0.00297399,0.0002633723,0.001533179,0.006296683,0.0002269485,0.000122403],"category_scores_gemma":[0.003214444,0.0004031502,0.0001393746,0.003038156,0.0008552475,0.0005053424,0.004124863,0.0007180666,0.0011085],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001986252,"about_ca_system_score_gemma":0.0003473034,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001125982,"about_ca_topic_score_gemma":0.00123081,"domain_scores_codex":[0.990858,0.00008122923,0.001356967,0.003241597,0.003597322,0.0008649232],"domain_scores_gemma":[0.9927951,0.002732856,0.0004606271,0.003572521,0.0002449852,0.0001938814],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001174702,0.00003027055,0.0005825456,0.00001388173,0.000004909231,0.0002481842,0.0002991995,0.05210276,0.00002730575,0.001860366,0.00646933,0.9383495],"study_design_scores_gemma":[0.000540477,0.00008993706,0.003875757,0.0005544776,0.000009340565,0.00002159914,0.000004842526,0.4858554,0.0001097155,0.4634443,0.04450845,0.0009856788],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0006206445,0.0001957527,0.982355,0.001346032,0.009235783,0.0005182301,0.000299726,0.0002078838,0.005220934],"genre_scores_gemma":[0.6791643,0.0003230644,0.248693,0.01034965,0.005491049,0.00007171518,0.001535483,0.0003800767,0.05399168],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9373638,"threshold_uncertainty_score":0.999842,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09353235581372939,"score_gpt":0.3594079730076309,"score_spread":0.2658756171939016,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}