{"id":"W4385570349","doi":"10.18653/v1/2023.findings-acl.215","title":"Varta: A Large-Scale Headline-Generation Dataset for Indic Languages","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; Canadian Institute for Advanced Research; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Headline; Computer science; Natural language processing; Artificial intelligence; Scale (ratio); Variety (cybernetics); Information retrieval; Data science; Linguistics; Geography","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000297796,0.00005616622,0.00006678705,0.00007224127,0.00008498538,0.00008030215,0.0003044218,0.0000314218,0.00002302494],"category_scores_gemma":[0.0000262998,0.00004881548,0.0000222032,0.0002039663,0.000004000747,0.0002362137,0.0001390424,0.00003359823,0.0001320148],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000009247543,"about_ca_system_score_gemma":0.00002330806,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001913414,"about_ca_topic_score_gemma":0.0001712679,"domain_scores_codex":[0.9993207,0.00001532644,0.0001143451,0.000242914,0.0001124739,0.0001941878],"domain_scores_gemma":[0.9995024,0.00003775435,0.00002164151,0.000379174,0.00001982343,0.00003919412],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00000431473,0.00009951679,0.0006616568,0.000062645,0.00002827553,0.00002494185,0.003294749,0.0046285,0.01181104,0.09616875,0.8185554,0.06466018],"study_design_scores_gemma":[0.0002347623,0.00001991431,0.0001600691,0.000002532562,0.000002142619,0.000002234792,0.00007038768,0.955593,0.002196434,0.000570021,0.04106533,0.00008317878],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01401482,0.00002394814,0.9829119,0.001939144,0.0002194927,0.0001453482,0.0003553488,0.000225335,0.0001646131],"genre_scores_gemma":[0.5294078,0.00001666841,0.4549727,0.003852655,0.001092819,0.0001481867,0.005390045,0.00002284294,0.005096308],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9509645,"threshold_uncertainty_score":0.1990637,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05068896009991682,"score_gpt":0.3264950599578167,"score_spread":0.2758060998578999,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}