{"id":"W3003877329","doi":"10.1145/1227504.1227324","title":"Plagiarism detection using feature-based neural networks","year":2007,"lang":"en","type":"article","venue":"ACM SIGCSE Bulletin","topic":"Academic integrity and plagiarism","field":"Social Sciences","cited_by":58,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Plagiarism detection; Feature (linguistics); Relevance (law); Computer science; Detector; Artificial neural network; Similarity (geometry); Artificial intelligence; Code (set theory); Feature engineering; Measure (data warehouse); Pattern recognition (psychology); Natural language processing; Machine learning; Data mining; Deep learning; Programming language; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["research_integrity"],"domain":null,"study_design":"bench_or_experimental","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"medium","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"design_other","genre":"methods","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","research_integrity"],"consensus_categories":[],"category_scores_codex":[0.001885177,0.000814249,0.0008185737,0.004873024,0.0006082486,0.001278125,0.001096605,0.000979326,0.001392761],"category_scores_gemma":[0.009988482,0.000276174,0.0004791775,0.00187419,0.0004314125,0.001670396,0.0007256036,0.000951722,0.0007720705],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001110694,"about_ca_system_score_gemma":0.0006490824,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005577264,"about_ca_topic_score_gemma":0.006424455,"domain_scores_codex":[0.9986791,0.0002913054,0.0001222086,0.0003269384,0.0004629309,0.0001175436],"domain_scores_gemma":[0.9910427,0.003957314,0.001623658,0.0005504876,0.002614347,0.0002114605],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0005672886,0.0004863467,0.02347699,0.0002109021,0.0001731189,0.0001442536,0.0001825343,0.04343354,0.02575909,0.0008392698,0.003671267,0.9010555],"study_design_scores_gemma":[0.0000298152,0.0001556512,0.01338706,0.00003288072,0.00005505476,0.0001108544,0.00005316113,0.9621273,0.02083555,0.002147625,0.001022814,0.00004225802],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4727884,0.001141657,0.5075353,0.0006596867,0.0001375554,0.0003818509,0.0009054962,0.0106445,0.005805519],"genre_scores_gemma":[0.8906468,0.0001887136,0.1061211,0.0000710973,0.00006466696,0.0001207185,0.0006403373,0.00006979401,0.002076906],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9990207,"threshold_uncertainty_score":0.01108962,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02276400814018628,"score_gpt":0.2937078578585832,"score_spread":0.270943849718397,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}