{"id":"W85461845","doi":"10.1007/978-3-642-54903-8_33","title":"How Document Properties Affect Document Relatedness Measures","year":2014,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Trigram; Computer science; Document classification; Quality (philosophy); Information retrieval; Vector space model; Natural language processing; Word (group theory); Property (philosophy); Affect (linguistics); Artificial intelligence; Space (punctuation); Weighting; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005212633,0.0005604284,0.0006433406,0.003638733,0.000861801,0.004201354,0.0005656409,0.0009375276,0.003562875],"category_scores_gemma":[0.05500298,0.0003477451,0.0006274694,0.004182603,0.0005477081,0.006512344,0.0008275325,0.001148818,0.00237315],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007319654,"about_ca_system_score_gemma":0.0004366668,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001336136,"about_ca_topic_score_gemma":0.002096633,"domain_scores_codex":[0.9957159,0.001771071,0.000352887,0.000824665,0.001135909,0.0001994732],"domain_scores_gemma":[0.9482888,0.04121174,0.002182024,0.002905943,0.004751066,0.0006604442],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002876518,0.000595161,0.1124243,0.002020556,0.0008126484,0.0003782616,0.002083159,0.0188939,0.1366657,0.01572372,0.02595935,0.6815667],"study_design_scores_gemma":[0.0002580009,0.002059073,0.283715,0.0006311614,0.003069496,0.003125351,0.00290388,0.3341367,0.2245031,0.08665848,0.0584651,0.0004747233],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"methods","genre_scores_codex":[0.785351,0.0114463,0.1696792,0.001117486,0.0005800091,0.0002495495,0.004707719,0.004112749,0.0227559],"genre_scores_gemma":[0.9405009,0.001387795,0.0487145,0.000136899,0.0002276912,0.00008075106,0.004424816,0.0008825583,0.003644101],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.005212633,"threshold_uncertainty_score":0.02756739,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01642841206557456,"score_gpt":0.2436465799015746,"score_spread":0.227218167836,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}