{"id":"W2964122685","doi":"10.18653/v1/p19-1153","title":"Towards Lossless Encoding of Sentences","year":2019,"lang":"","type":"article","venue":"PolyPublie (École Polytechnique de Montréal)","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Lossless compression; Computer science; Sentence; Encoding (memory); Embedding; Feature (linguistics); Natural language processing; Focus (optics); Artificial intelligence; Task (project management); Sequence labeling; Compression (physics); Sequence (biology); Data compression; Speech recognition; Linguistics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008398527,0.0008026027,0.0005949671,0.000820921,0.0002304722,0.0009343583,0.0007246296,0.0007121551,0.003045989],"category_scores_gemma":[0.005924236,0.000261421,0.0004453466,0.000710211,0.0005892169,0.003109876,0.001429162,0.001545349,0.002356959],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004319917,"about_ca_system_score_gemma":0.0005744572,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005488197,"about_ca_topic_score_gemma":0.0008942449,"domain_scores_codex":[0.9990742,0.0003459723,0.00007635703,0.0001690856,0.0002585385,0.00007580857],"domain_scores_gemma":[0.9982326,0.0006555055,0.0001476983,0.0005391542,0.0003596474,0.00006539841],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0007578876,0.0002156046,0.0007601039,0.0003993174,0.00006658432,0.000349866,0.0005501464,0.07878282,0.06074939,0.07601732,0.03502429,0.7463267],"study_design_scores_gemma":[0.00006424676,0.0003266776,0.0004798994,0.00008949218,0.00004474722,0.0003809097,0.0001709728,0.8040449,0.0503252,0.1139009,0.03013169,0.00004036318],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02108506,0.00071178,0.9721596,0.0006363914,0.0002263237,0.00005180041,0.000813822,0.002320851,0.001994456],"genre_scores_gemma":[0.35621,0.001639148,0.6191216,0.001312909,0.0006010467,0.0004543932,0.005256681,0.001076182,0.01432812],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003045989,"threshold_uncertainty_score":0.01018983,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0172286681527959,"score_gpt":0.2410758530218186,"score_spread":0.2238471848690227,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}