{"id":"W4414153271","doi":"10.1016/j.eswa.2025.129655","title":"The power of text similarity in identifying AI-LLM paraphrased documents: The case of BBC news articles and ChatGPT","year":2025,"lang":"en","type":"article","venue":"Expert Systems with Applications","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Similarity (geometry); Task (project management); Benchmark (surveying); Generative grammar; Power (physics); Revenue","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003613648,0.00008018809,0.0001372838,0.00006638284,0.0002065541,0.000106775,0.0004276376,0.00003081253,7.819158e-7],"category_scores_gemma":[0.0000195195,0.00004611681,0.00002218413,0.0004705449,0.00009832885,0.0001618097,0.0001200467,0.00007829301,8.433922e-7],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002306744,"about_ca_system_score_gemma":0.00005271251,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002228726,"about_ca_topic_score_gemma":0.0006852907,"domain_scores_codex":[0.9990759,0.00008750627,0.00035608,0.000226411,0.0001205713,0.0001335526],"domain_scores_gemma":[0.9986694,0.0002412637,0.000132823,0.0008487491,0.00007953877,0.00002821183],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00003033022,0.000342806,0.01163505,0.0002101174,0.0001432214,0.00002413297,0.02951716,0.001782837,0.004830417,0.9161867,0.001358367,0.03393884],"study_design_scores_gemma":[0.005186399,0.0002407466,0.01015954,0.001633263,0.00008871585,0.0006059711,0.0987235,0.7694348,0.02028216,0.04093248,0.0515193,0.00119316],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1171883,0.003741806,0.8738314,0.003557434,0.00008363334,0.001098433,0.000002486621,0.00003139424,0.0004651177],"genre_scores_gemma":[0.9973218,0.00004061796,0.001864587,0.0001422782,0.00001095264,0.000548992,2.880788e-7,0.00000349022,0.00006696642],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8801335,"threshold_uncertainty_score":0.3369182,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01692661947228411,"score_gpt":0.2993149508804233,"score_spread":0.2823883314081391,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}