{"id":"W1989001126","doi":"10.3115/1220355.1220502","title":"Fast computation of lexical affinity models","year":2004,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"Coordenação de Aperfeiçoamento de Pessoal de Nível Superior","keywords":"Computer science; Computation; Scalability; Terabyte; Independence (probability theory); Artificial intelligence; Parametric statistics; Similarity (geometry); Focus (optics); Natural language processing; Theoretical computer science; Algorithm; Mathematics; Database","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00008870302,0.00005158138,0.00007597683,0.00004954368,0.00002632588,0.00003305637,0.0003552538,0.00003780142,0.000001779037],"category_scores_gemma":[0.00001282645,0.00004130817,0.00002549505,0.0001940058,0.00002809632,0.0004442908,0.0001215816,0.00006782213,0.000004121164],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002152916,"about_ca_system_score_gemma":0.0000432684,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0000641408,"about_ca_topic_score_gemma":0.000005363142,"domain_scores_codex":[0.9995033,0.00001120908,0.0001153748,0.000133644,0.0001504076,0.00008604057],"domain_scores_gemma":[0.9996975,0.00001380065,0.00004378103,0.0001430211,0.00007396289,0.00002794135],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00000177153,0.00004981473,0.000005985489,0.0000117403,0.000002105835,0.00000300052,0.0003077672,0.004404555,0.002096643,0.9617724,0.00004933486,0.03129485],"study_design_scores_gemma":[0.0001196639,0.00003991116,0.00002666734,0.00001960695,9.274959e-7,0.000006693706,0.000006155493,0.07873671,0.09822882,0.8227472,0.000002368218,0.00006532301],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.005796278,0.0001316919,0.9909024,0.0004705239,0.00003089308,0.00004482071,3.028941e-7,0.0004028879,0.002220224],"genre_scores_gemma":[0.518683,6.583738e-7,0.4812221,0.00007739612,0.000004689123,7.3818e-7,3.42666e-7,0.000001289383,0.00000984404],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.5128866,"threshold_uncertainty_score":0.1684498,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02419785949263349,"score_gpt":0.2846610855098097,"score_spread":0.2604632260171763,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}