{"id":"W2589404650","doi":"10.29173/cais437","title":"Statistical Power and Effect Size in Informative Retrieval Experiments","year":2013,"lang":"en","type":"article","venue":"Proceedings of the Annual Conference of CAIS / Actes du congrès annuel de l ACSI","topic":"Information Retrieval and Search Behavior","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Western University","funders":"","keywords":"Statistical hypothesis testing; Null hypothesis; Statistical power; Type I and type II errors; Null (SQL); Alternative hypothesis; Statistical analysis; Search engine indexing; Computer science; Statistics; Multiple comparisons problem; Mathematics; Artificial intelligence; Information retrieval; Data mining","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2351994,0.00221812,0.004089441,0.004234126,0.00174479,0.004545408,0.003389078,0.008076949,0.0119403],"category_scores_gemma":[0.6786096,0.001489508,0.003904953,0.004006585,0.01266611,0.01120013,0.004871073,0.006519475,0.00167958],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001912063,"about_ca_system_score_gemma":0.001809823,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004640427,"about_ca_topic_score_gemma":0.0002613577,"domain_scores_codex":[0.7241271,0.1922771,0.01339419,0.0234792,0.04417615,0.002546185],"domain_scores_gemma":[0.1547717,0.7940466,0.01738016,0.0267353,0.005342694,0.001723601],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.08662762,0.007387021,0.1071061,0.01294105,0.01013222,0.003282999,0.008268393,0.03601798,0.04551016,0.3678213,0.01696704,0.2979381],"study_design_scores_gemma":[0.01583522,0.02837794,0.1818125,0.001347021,0.004647276,0.001934516,0.001062633,0.05404206,0.02472237,0.6632498,0.02223512,0.0007336546],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3092928,0.009288673,0.5672752,0.008910644,0.004516688,0.02065679,0.004721745,0.001383919,0.07395355],"genre_scores_gemma":[0.8448245,0.000798991,0.1104618,0.003407605,0.001069278,0.0340809,0.0008781176,0.0006483936,0.003830445],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7648005,"threshold_uncertainty_score":0.9431353,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01377638208287326,"score_gpt":0.2600525226492694,"score_spread":0.2462761405663962,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}