{"id":"W4402422143","doi":"10.1515/lingvan-2023-0102","title":"Bibliographic bias and information-density sampling","year":2024,"lang":"en","type":"article","venue":"Linguistics Vanguard","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"Marcus och Amalia Wallenbergs minnesfond","keywords":"Sampling bias; Sampling (signal processing); Statistics; Information retrieval; Computer science; Geography; Mathematics; Sample size determination; Telecommunications","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.2357713,0.0006860632,0.002780643,0.02451419,0.004284187,0.009370446,0.004950193,0.002544479,0.006182659],"category_scores_gemma":[0.6992773,0.00117837,0.001006363,0.04074041,0.01157787,0.01206647,0.007245197,0.002208245,0.001210379],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005876318,"about_ca_system_score_gemma":0.003709164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006922584,"about_ca_topic_score_gemma":0.005930678,"domain_scores_codex":[0.6123956,0.2940697,0.0239575,0.01956209,0.04731511,0.002699922],"domain_scores_gemma":[0.1295011,0.7637253,0.03291723,0.04990686,0.02281991,0.001129556],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006081325,0.0001675689,0.1112997,0.004715011,0.001001008,0.001227365,0.02224816,0.007240648,0.001059244,0.5850279,0.01309213,0.2523131],"study_design_scores_gemma":[0.0002543362,0.0001829965,0.04888704,0.00485725,0.0005624277,0.002268831,0.01222187,0.02936131,0.00294821,0.8314807,0.06675147,0.0002234393],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2159551,0.01712636,0.6698187,0.02017785,0.001079099,0.00262104,0.0038725,0.0009101343,0.06843914],"genre_scores_gemma":[0.8796324,0.002908094,0.1076712,0.002133375,0.0008125696,0.003153472,0.001341766,0.000149201,0.002197946],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9754858,"threshold_uncertainty_score":0.9424301,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02514485621604285,"score_gpt":0.2974952076023429,"score_spread":0.2723503513863001,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}