{"id":"W7066035670","doi":"","title":"False Positives: Opportunities and Dangers in Big Text Analysis","year":2012,"lang":"en","type":"other","venue":"KU ScholarWorks (The University of Kansas)","topic":"Historical and Linguistic Studies","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Big data; Digital humanities; Data collection; Information technology; Field (mathematics); Digital forensics","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0005174034,0.0001510647,0.0004071408,0.0004558891,0.0004856028,0.00002069913,0.0003463591,0.0002596693,0.001501608],"category_scores_gemma":[0.0001660383,0.000137598,0.0001415417,0.0006653751,0.001007853,0.00005676666,0.00009678188,0.0003096762,0.00002499538],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001228675,"about_ca_system_score_gemma":0.00008028333,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.01968986,"about_ca_topic_score_gemma":0.02471375,"domain_scores_codex":[0.9988386,0.0002898326,0.0001107435,0.0002115081,0.0002898416,0.0002594911],"domain_scores_gemma":[0.9992333,0.0002162897,0.0001689041,0.0001870212,0.00006204817,0.0001324246],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001523964,0.0005177121,0.07310015,0.0002273731,0.005478521,0.0002586736,0.3475021,0.00002233802,0.000004981887,0.06762991,0.3480106,0.1570953],"study_design_scores_gemma":[0.0001378873,0.000008428042,0.003904452,0.00008793646,0.0006042791,1.206004e-7,0.03902047,0.000002336255,1.057466e-7,0.00004430534,0.9560268,0.0001629123],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.0006066149,0.007996087,0.0002812303,0.0009418018,0.0002989697,0.0001735284,0.00004482771,0.00005223303,0.9896047],"genre_scores_gemma":[0.1067843,0.007867264,0.0002889703,0.00008518952,0.0003977946,5.002782e-7,0.00001113071,0.00004319377,0.8845217],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.6080162,"threshold_uncertainty_score":0.9994112,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03779193104012701,"score_gpt":0.2470813842919597,"score_spread":0.2092894532518327,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}