{"id":"W4401666276","doi":"10.1007/s10618-024-01054-7","title":"Bayesian network Motifs for reasoning over heterogeneous unlinked datasets","year":2024,"lang":"en","type":"article","venue":"Data Mining and Knowledge Discovery","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Canada Research Chairs; University of Toronto; University of New Brunswick","funders":"","keywords":"Computer science; Probabilistic logic; Bayesian network; Python (programming language); Data mining; Aggregate (composite); Real world data; Bayesian probability; Machine learning; Set (abstract data type); Artificial intelligence; Data science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006303884,0.0009469164,0.001724678,0.005379261,0.001370603,0.003377518,0.003754775,0.003106374,0.005181895],"category_scores_gemma":[0.06944677,0.001202293,0.002209681,0.004837027,0.001509382,0.01009097,0.00346355,0.003721661,0.0009479563],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002118853,"about_ca_system_score_gemma":0.002374718,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009635517,"about_ca_topic_score_gemma":0.01450601,"domain_scores_codex":[0.9952177,0.001625648,0.0004265507,0.001037123,0.001462882,0.0002300921],"domain_scores_gemma":[0.9605373,0.03230933,0.002391583,0.002392633,0.001678909,0.0006903972],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006342022,0.0004632248,0.01036486,0.001027713,0.0007022623,0.001006005,0.0008345693,0.3029706,0.003286767,0.3915555,0.00920718,0.2779472],"study_design_scores_gemma":[0.00003602022,0.00002429004,0.0003302174,0.00007642182,0.00007483584,0.000129336,0.00006295311,0.6005837,0.000652386,0.3958277,0.002185132,0.00001702007],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01265743,0.0005827165,0.9825656,0.000777947,0.00004645783,0.000131952,0.001038169,0.001276814,0.0009229041],"genre_scores_gemma":[0.270836,0.000897358,0.7210087,0.0004082642,0.0001431424,0.0004469147,0.003841622,0.0002950105,0.002122839],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.009635517,"threshold_uncertainty_score":0.03333849,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1674946705782729,"score_gpt":0.4358667777938899,"score_spread":0.2683721072156171,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}