{"id":"W3021034694","doi":"10.18653/v1/2020.acl-main.695","title":"The Paradigm Discovery Problem","year":2020,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"York University; New York University Abu Dhabi","keywords":"Computer science; Benchmark (surveying); Task (project management); Cluster analysis; String (physics); Heuristic; Construct (python library); Artificial intelligence; Word (group theory); Code (set theory); Natural language processing; Machine learning; Programming language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0001821411,0.0001454575,0.0001385388,0.0000152847,0.0001153268,0.001147534,0.002431449,0.00008134715,0.000002772562],"category_scores_gemma":[0.00001728314,0.00008751905,0.00009581365,0.00006721531,0.00002337911,0.0002021479,0.003715039,0.0004280425,0.00005876527],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00003208168,"about_ca_system_score_gemma":0.0001887179,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007158862,"about_ca_topic_score_gemma":0.00001832701,"domain_scores_codex":[0.9987331,0.0000445447,0.0002222164,0.0005441995,0.0002478857,0.0002080249],"domain_scores_gemma":[0.9985569,0.00008884813,0.0000848914,0.001195201,0.00001515774,0.00005896484],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[6.896324e-7,0.000004741212,0.00002641993,0.00002289004,0.00001631933,0.0000052769,0.0002389014,0.002194076,0.000006060509,0.9767987,0.00254435,0.01814158],"study_design_scores_gemma":[0.00004180343,0.000007639597,0.00006926051,0.00002547758,0.000003465179,0.000002551085,0.000006942174,0.3778881,0.00007252098,0.5977031,0.02400777,0.0001713244],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.000151657,0.0002414879,0.9399917,0.03785048,0.0007653895,0.0002403337,9.56278e-7,0.0002531147,0.02050494],"genre_scores_gemma":[0.5793813,0.0002668778,0.4065831,0.002895137,0.0007948389,0.0001730527,0.000005882638,0.00002968011,0.009870118],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.5792297,"threshold_uncertainty_score":0.9998894,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03907413907118272,"score_gpt":0.2561360138813221,"score_spread":0.2170618748101393,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}