{"id":"W2742155240","doi":"10.18653/v1/w17-2512","title":"Overview of the Second BUCC Shared Task: Spotting Parallel Sentences in Comparable Corpora","year":2017,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"ca_institutions":"Canadian Nautical Research Society","funders":"European Commission","keywords":"Sentence; Computer science; German; Task (project management); Natural language processing; Artificial intelligence; Parallel corpora; Gold standard (test); Spotting; Sample (material); Speech recognition; Linguistics; Statistics; Mathematics; Machine translation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02064348,0.006551602,0.005794964,0.01374497,0.005774793,0.007075632,0.008214633,0.005278572,0.02237502],"category_scores_gemma":[0.04217049,0.002505239,0.004238118,0.008373969,0.002308825,0.008519266,0.01279348,0.007732494,0.02965881],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00400853,"about_ca_system_score_gemma":0.01169495,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02269433,"about_ca_topic_score_gemma":0.02722715,"domain_scores_codex":[0.9759168,0.007446389,0.002512222,0.006645875,0.005772847,0.001705902],"domain_scores_gemma":[0.9588106,0.007658236,0.001119283,0.01608398,0.01370379,0.002624064],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001267345,0.001358744,0.004316269,0.004230449,0.0009860911,0.0007559671,0.001294855,0.007403273,0.04239356,0.004568111,0.5255759,0.4058495],"study_design_scores_gemma":[0.001654974,0.002456453,0.02873482,0.00111872,0.0008823472,0.003966978,0.002206533,0.1125193,0.0888447,0.01629502,0.7400359,0.001284249],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06616517,0.01126162,0.4810588,0.004064674,0.004042901,0.01498006,0.1784762,0.2017822,0.0381683],"genre_scores_gemma":[0.04043092,0.0009584395,0.3776363,0.001014329,0.000771874,0.01114767,0.5458996,0.01139138,0.01074951],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.02269433,"threshold_uncertainty_score":0.1091744,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05596415462690715,"score_gpt":0.3149779532429419,"score_spread":0.2590137986160347,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}