{"id":"W3100889051","doi":"10.48550/arxiv.2011.05723","title":"CalibreNet: Calibration Networks for Multilingual Sequence Labeling","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Simon Fraser University","funders":"People's Government of Jilin Province; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Computer science; Benchmark (surveying); Sequence labeling; Task (project management); Sequence (biology); Named-entity recognition; Phrase; Natural language processing; Artificial intelligence; Obstacle; Boundary (topology); Resource (disambiguation); Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.0002120156,0.0003028522,0.0003194008,0.0001060454,0.0001727105,0.0001918879,0.001691002,0.0003473777,0.000005020124],"category_scores_gemma":[0.00007289873,0.0003755212,0.0001918793,0.0003289063,0.00004819162,0.0004109452,0.001502029,0.0005276536,0.000006504039],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000152302,"about_ca_system_score_gemma":0.0002913458,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001607314,"about_ca_topic_score_gemma":0.00004060183,"domain_scores_codex":[0.9977807,0.00009434311,0.0002705693,0.001380327,0.00008827449,0.0003857322],"domain_scores_gemma":[0.9983312,0.0001565055,0.0002417865,0.0009314331,0.0001436547,0.0001954173],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001446412,0.00001440082,0.0002779724,0.00005984391,0.00003265813,0.00006498811,0.0002248491,0.9331521,0.00009183467,0.06413208,0.00006535224,0.001869506],"study_design_scores_gemma":[0.0003358886,0.00002754132,0.00001585697,0.00006842182,0.00003487203,0.000001341232,0.00002415683,0.972993,0.0002476996,0.02567597,0.0002070963,0.0003681463],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.04027451,0.00008045955,0.9574564,0.0002927179,0.0008833528,0.0004566604,0.00002053269,0.0004321146,0.0001033141],"genre_scores_gemma":[0.9560183,0.00005065602,0.04299398,0.0003262074,0.0003308217,0.000002769477,0.00004563124,0.00002483235,0.0002067647],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9157438,"threshold_uncertainty_score":0.9998696,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1565945791399868,"score_gpt":0.2257898305062738,"score_spread":0.06919525136628696,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}