{"id":"W4401267042","doi":"10.2139/ssrn.4885662","title":"Will User-Contributed AI Training Data Eat Its Own Tail?","year":2024,"lang":"en","type":"preprint","venue":"SSRN Electronic Journal","topic":"Explainable Artificial Intelligence (XAI)","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Training (meteorology); Computer science; Geography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","open_science","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.006266422,0.0006892885,0.0007522017,0.0005674651,0.0004977535,0.002178887,0.009453242,0.0005159613,0.00005730444],"category_scores_gemma":[0.0005840783,0.0006370381,0.0003285459,0.0008496387,0.00008508608,0.002310864,0.007597149,0.01351347,0.0005736255],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00223845,"about_ca_system_score_gemma":0.01424902,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001356756,"about_ca_topic_score_gemma":0.001934307,"domain_scores_codex":[0.9904336,0.0003489566,0.001099057,0.001715215,0.001029884,0.005373283],"domain_scores_gemma":[0.9957209,0.0001945497,0.0005266253,0.002706268,0.0005110013,0.0003407101],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00003307951,0.00008034956,0.00002226499,0.00007102753,0.0006794033,0.0002878648,0.001592433,0.002119281,0.0004143595,0.8900645,0.00269678,0.1019386],"study_design_scores_gemma":[0.0001920396,0.0002322342,0.000004498208,0.0003554919,0.0001162051,0.001113289,0.0009121217,0.1074489,0.0007978369,0.8569556,0.0310979,0.0007739337],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01271247,0.03464816,0.9225699,0.02190989,0.005509316,0.0007500848,0.0001260755,0.0007458376,0.001028322],"genre_scores_gemma":[0.9748918,0.01107373,0.003465934,0.001071344,0.002890871,0.000044404,0.0001518413,0.0001587668,0.006251242],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.9621794,"threshold_uncertainty_score":0.9996081,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05195564040774149,"score_gpt":0.3158387788508273,"score_spread":0.2638831384430858,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}