{"id":"W2115008552","doi":"10.5539/mas.v4n9p142","title":"Entropy Based Measurement of Text Dissimilarity for Duplicate – Detection","year":2010,"lang":"en","type":"article","venue":"Modern Applied Science","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Similarity (geometry); Entropy (arrow of time); Data mining; Pattern recognition (psychology); Artificial intelligence; Physics; Image (mathematics)","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002638183,0.0004696979,0.001071834,0.006733059,0.0005185375,0.001087841,0.0008962678,0.0007757664,0.001506725],"category_scores_gemma":[0.01630239,0.0001643769,0.0005894447,0.005014434,0.0007509044,0.002530274,0.001360909,0.0006332286,0.0006291518],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005252,"about_ca_system_score_gemma":0.0004236672,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0003361109,"about_ca_topic_score_gemma":0.0003818193,"domain_scores_codex":[0.9956455,0.0008597717,0.0005683671,0.0005643174,0.002199639,0.0001623415],"domain_scores_gemma":[0.988959,0.00537318,0.001660319,0.001642668,0.002091831,0.000273166],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001720813,0.0004592988,0.06767561,0.001297306,0.0005575246,0.0009735441,0.001115043,0.04111168,0.1650503,0.02001567,0.004924445,0.6950989],"study_design_scores_gemma":[0.00008505856,0.001249652,0.1319598,0.0001262074,0.0002967229,0.004495255,0.000766948,0.639249,0.1703917,0.03934325,0.01167451,0.0003617977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.3071664,0.001800879,0.6809821,0.0002936705,0.0002640336,0.0003270328,0.002272899,0.001330087,0.005562995],"genre_scores_gemma":[0.8089855,0.000385598,0.187077,0.0000616805,0.0001927074,0.00018006,0.00187946,0.00008882501,0.001149147],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006733059,"threshold_uncertainty_score":0.0139522,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1460390288590918,"score_gpt":0.3735185434045063,"score_spread":0.2274795145454145,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}