{"id":"W2949034987","doi":"10.18653/v1/p19-1153","title":"Towards Lossless Encoding of Sentences","year":2019,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Compute Canada","keywords":"Lossless compression; Sentence; Computer science; Encoding (memory); Embedding; Feature (linguistics); Natural language processing; Focus (optics); Task (project management); Artificial intelligence; Sequence labeling; Compression (physics); Sequence (biology); Field (mathematics); Speech recognition; Data compression; Linguistics; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008791906,0.0008459952,0.0006156294,0.0008902174,0.0002536892,0.001026038,0.0007866083,0.0007557624,0.003231573],"category_scores_gemma":[0.006543089,0.0003003562,0.0004917859,0.0007848727,0.0006480952,0.003371688,0.001575727,0.001769307,0.002389703],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004514326,"about_ca_system_score_gemma":0.0005814673,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005305022,"about_ca_topic_score_gemma":0.0007980874,"domain_scores_codex":[0.9989933,0.0003778544,0.00008120037,0.0001965459,0.0002716742,0.00007945277],"domain_scores_gemma":[0.9980242,0.0007672873,0.0001617119,0.0006085351,0.0003665841,0.0000717584],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0007578785,0.0002163215,0.0007919,0.0004391605,0.00007325368,0.0003683021,0.0005945611,0.08940563,0.05965897,0.09561447,0.03360361,0.718476],"study_design_scores_gemma":[0.00005992281,0.0002893527,0.000439606,0.00009232789,0.000043826,0.0003653482,0.0001679055,0.7894003,0.04070337,0.1413424,0.02705451,0.00004119413],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01880428,0.0006285949,0.9750491,0.0006135132,0.000209025,0.00004691114,0.0007794959,0.002034744,0.001834399],"genre_scores_gemma":[0.344735,0.001677547,0.6302925,0.001270427,0.0006541968,0.0004817166,0.005493929,0.001200647,0.01419405],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.003231573,"threshold_uncertainty_score":0.01081073,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05661246926732516,"score_gpt":0.288773762604461,"score_spread":0.2321612933371358,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}