{"id":"W1983388548","doi":"10.1007/s10844-012-0229-0","title":"Reducing the size of databases for multirelational classification: a subgraph-based approach","year":2012,"lang":"en","type":"article","venue":"Journal of Intelligent Information Systems","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Ottawa; National Research Council Canada","funders":"","keywords":"Computer science; Database; Preprocessor; Tuple; Data mining; Relational database; Data pre-processing; Schema (genetic algorithms); Machine learning; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001648178,0.0009828124,0.002760458,0.004511752,0.001379269,0.002483575,0.003293507,0.001371466,0.003001158],"category_scores_gemma":[0.01294419,0.0008008998,0.002673316,0.005991275,0.001020151,0.005780153,0.002671729,0.001320185,0.0008374558],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001296954,"about_ca_system_score_gemma":0.002605044,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01057202,"about_ca_topic_score_gemma":0.01939943,"domain_scores_codex":[0.9970862,0.0009115334,0.0002560114,0.0005691987,0.0008861643,0.00029087],"domain_scores_gemma":[0.9864702,0.005950574,0.0007711134,0.00442918,0.001947329,0.0004316931],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007949385,0.000726792,0.00951367,0.000914754,0.0005427439,0.0004489266,0.0007295593,0.1846844,0.03062847,0.03792151,0.01816648,0.7149278],"study_design_scores_gemma":[0.00007123321,0.0001626064,0.00237532,0.00005448109,0.0002997083,0.0002975422,0.0004904006,0.9083661,0.006351848,0.07638277,0.005107142,0.00004090392],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.07274996,0.001019298,0.9175188,0.001373487,0.0001058594,0.0003395637,0.001455453,0.003044857,0.002392752],"genre_scores_gemma":[0.3389491,0.0006730571,0.6530453,0.0003168546,0.0001238166,0.0002368763,0.00390146,0.0005650625,0.002188371],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01057202,"threshold_uncertainty_score":0.02102095,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3453831662634866,"score_gpt":0.421210654523675,"score_spread":0.07582748826018837,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}