{"id":"W2953243682","doi":"10.48550/arxiv.1012.5011","title":"Towards a theoretical understanding of false positives in DNA motif finding","year":2010,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Genomics and Chromatin Dynamics","field":"Biochemistry, Genetics and Molecular Biology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"False positive paradox; True positive rate; Motif (music); False positives and false negatives; Rule of thumb; False positive rate; Computer science; Sequence motif; Pattern recognition (psychology); Artificial intelligence; Mathematics; Algorithm; Biology; DNA; Genetics; Physics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06471237,0.002876705,0.00365611,0.007123365,0.00247829,0.01037199,0.008629765,0.007961366,0.0045884],"category_scores_gemma":[0.3795999,0.002856717,0.003432364,0.003833856,0.01538414,0.01942433,0.007836591,0.01251892,0.001153725],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005207488,"about_ca_system_score_gemma":0.002955064,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001574354,"about_ca_topic_score_gemma":0.001035289,"domain_scores_codex":[0.9598154,0.02048763,0.001846108,0.005482884,0.01060215,0.001765807],"domain_scores_gemma":[0.3852751,0.5735659,0.01107864,0.01796776,0.01040438,0.001708235],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0002975073,0.0002492354,0.01004839,0.001138342,0.000258817,0.001422043,0.001172961,0.2098106,0.004028598,0.7263749,0.005084661,0.04011396],"study_design_scores_gemma":[0.00004172935,0.00007748327,0.0006675333,0.0002039468,0.00004374337,0.0006257072,0.00008216389,0.5085337,0.001907555,0.4868466,0.0009039127,0.00006591894],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01839888,0.001319193,0.9714546,0.003890942,0.0001740333,0.00008690979,0.0001521383,0.000494888,0.004028497],"genre_scores_gemma":[0.5238521,0.003583373,0.4595485,0.004700244,0.001500884,0.001371906,0.0007741746,0.0008284046,0.003840556],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.06471237,"threshold_uncertainty_score":0.3422358,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03868726073869447,"score_gpt":0.1969264299763184,"score_spread":0.1582391692376239,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}