{"id":"W3196643997","doi":"10.18653/v1/2021.findings-emnlp.203","title":"Refining BERT Embeddings for Document Hashing via Mutual Information Maximization","year":2021,"lang":"en","type":"preprint","venue":"","topic":"Advanced Image and Video Retrieval Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Natural Science Foundation of Guangdong Province; National Natural Science Foundation of China","keywords":"Computer science; Hash function; Benchmark (surveying); Mutual information; Generative grammar; Generative model; Maximization; Artificial intelligence; Machine learning; Data mining; Theoretical computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0004524405,0.0002571199,0.0002882683,0.0001990643,0.000157206,0.001296927,0.0007776627,0.000233851,0.0000200023],"category_scores_gemma":[0.0002642146,0.0002549618,0.0001501006,0.0002354865,0.00001727452,0.004173944,0.001625659,0.0003330568,0.000009565469],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001909715,"about_ca_system_score_gemma":0.0001370987,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00005350819,"about_ca_topic_score_gemma":0.000005472076,"domain_scores_codex":[0.9983266,0.00003185079,0.0005586089,0.0004521324,0.0003448398,0.0002859687],"domain_scores_gemma":[0.9983469,0.0001024644,0.0003830697,0.0006483878,0.000449162,0.00007003302],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001932725,0.00002847773,0.00003879781,0.0003435751,0.00004997863,0.000004478013,0.002451224,0.002341283,0.0006794601,0.02220594,0.002255739,0.9695817],"study_design_scores_gemma":[0.0009252431,0.000308405,0.0001284768,0.0009344175,0.0000606853,0.00004295543,0.0002820937,0.6032656,0.2377328,0.071734,0.0828238,0.001761566],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0002514934,0.00008219595,0.9954338,0.0005449987,0.0005115363,0.0004033763,0.000003615539,0.0007402682,0.002028703],"genre_scores_gemma":[0.02851756,0.00007960985,0.969351,0.0009956968,0.00008456533,0.0001719318,0.0002859566,0.00001582331,0.000497853],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9678202,"threshold_uncertainty_score":0.9999903,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01598472026073924,"score_gpt":0.291538124228188,"score_spread":0.2755534039674488,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}