{"id":"W4388976713","doi":"10.1007/978-3-031-45043-3_2","title":"Building an NLIDB: The Basics","year":2023,"lang":"en","type":"book-chapter","venue":"Synthesis lectures on data management","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Leverage (statistics); Sect; Construct (python library); Computer science; Baseline (sea); Rest (music); Programming language; Artificial intelligence; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002735528,0.0007780897,0.000956052,0.001770553,0.001862792,0.007484994,0.00229488,0.001320214,0.0587999],"category_scores_gemma":[0.007882097,0.00117269,0.001125681,0.002739548,0.001496018,0.01243393,0.005687674,0.00268411,0.06308035],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001515962,"about_ca_system_score_gemma":0.002016144,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004655389,"about_ca_topic_score_gemma":0.004239787,"domain_scores_codex":[0.9981888,0.0004258215,0.0001517108,0.0003672204,0.0007419334,0.0001245407],"domain_scores_gemma":[0.998193,0.0006523421,0.00004967262,0.0005298953,0.0004203503,0.0001547987],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00007674133,0.00008820785,0.0006063221,0.0005595636,0.0000197953,0.0001025547,0.001250068,0.002674111,0.004728707,0.2082024,0.1457696,0.6359218],"study_design_scores_gemma":[0.00001474467,0.00001587817,0.0002125669,0.0002186863,0.00001538764,0.0001872736,0.0003154099,0.01620337,0.004514454,0.1244581,0.8538142,0.00002991915],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.001487444,0.001613073,0.8979473,0.00253144,0.0005299159,0.0002290285,0.001947026,0.01448114,0.07923369],"genre_scores_gemma":[0.01722021,0.002088761,0.8920112,0.0008683563,0.0003133646,0.0005446728,0.006515993,0.007068194,0.07336929],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.0587999,"threshold_uncertainty_score":0.1967053,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09428031179532508,"score_gpt":0.2902417105555458,"score_spread":0.1959613987602207,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}