{"id":"W3005998127","doi":"10.22148/16.034","title":"A BLAST-based, Language-agnostic Text Reuse Algorithm with a MARKUS Implementation and Sequence Alignment Optimized for Large Chinese Corpora","year":2019,"lang":"en","type":"article","venue":"Journal of Cultural Analytics","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Reuse; Natural language processing; Phrase; Sequence (biology); Artificial intelligence; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002832057,0.002866157,0.002179322,0.002974736,0.003108259,0.002276474,0.002799159,0.001807993,0.01293735],"category_scores_gemma":[0.007849076,0.001697958,0.00191081,0.005937689,0.0009527617,0.002432511,0.002236833,0.003262169,0.01823928],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001165234,"about_ca_system_score_gemma":0.003592986,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004480249,"about_ca_topic_score_gemma":0.01347598,"domain_scores_codex":[0.998014,0.0003495892,0.0002669621,0.0008323491,0.0003758373,0.0001613102],"domain_scores_gemma":[0.9983144,0.0005388206,0.0001693907,0.0004022489,0.0004556921,0.0001195431],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004164942,0.0008601502,0.0111502,0.004060322,0.001176163,0.001953809,0.002877934,0.01010929,0.3003986,0.01350539,0.1473284,0.5024148],"study_design_scores_gemma":[0.001425735,0.001782929,0.02188251,0.0004686585,0.001382371,0.00553958,0.00209577,0.3627977,0.3089742,0.02960248,0.2633708,0.0006772167],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07497234,0.001228422,0.7125255,0.0005807053,0.0006907873,0.001498045,0.01887269,0.1793929,0.01023872],"genre_scores_gemma":[0.06024638,0.0003611268,0.8932496,0.0002877522,0.00006942928,0.001260037,0.02999623,0.008391843,0.006137557],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01293735,"threshold_uncertainty_score":0.04327983,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01158736095800556,"score_gpt":0.3090124364441694,"score_spread":0.2974250754861639,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}