{"id":"W3122916539","doi":"","title":"Mind the Gap: Accounting for Measurement Error and Misclassification in Variables Generated via Data Mining","year":2017,"lang":"en","type":"article","venue":"SSRN Electronic Journal","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Quest University Canada","funders":"","keywords":"Econometrics; Computer science; Econometric model; Covariate; Data mining; Errors-in-variables models; Observational error; Variance (accounting); Inference; Parameterized complexity; Statistics; Machine learning; Artificial intelligence; Algorithm; Mathematics; Accounting; Economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.130218,0.001703449,0.001909801,0.005841204,0.00220299,0.005157178,0.00488833,0.003991465,0.001405491],"category_scores_gemma":[0.5256914,0.0009274043,0.002448989,0.007529268,0.004928608,0.01024119,0.005514044,0.005136286,0.0004429988],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002796799,"about_ca_system_score_gemma":0.00452237,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008540502,"about_ca_topic_score_gemma":0.006147786,"domain_scores_codex":[0.8789345,0.08683025,0.008519769,0.01209489,0.0122571,0.001363482],"domain_scores_gemma":[0.4402853,0.4649687,0.03055924,0.0424468,0.02076156,0.0009783471],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004583258,0.0002070426,0.2374271,0.002097576,0.002860451,0.0007399239,0.009070124,0.07630512,0.0006446602,0.2909547,0.01602611,0.3632089],"study_design_scores_gemma":[0.0001236861,0.0002932452,0.04709387,0.003489946,0.0008658923,0.0008057123,0.002643621,0.3360136,0.0032384,0.5716559,0.0335281,0.0002481442],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.04910024,0.006015672,0.9257731,0.0106568,0.001346191,0.0004707178,0.0007623681,0.0005544704,0.005320436],"genre_scores_gemma":[0.6758251,0.002492702,0.3110394,0.005172152,0.001312683,0.001210468,0.001024554,0.0002745925,0.00164841],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.130218,"threshold_uncertainty_score":0.6886669,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09644917981351636,"score_gpt":0.331303410127973,"score_spread":0.2348542303144567,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}