{"id":"W4389519299","doi":"10.18653/v1/2023.nllp-1.24","title":"AsyLex: A Dataset for Legal Language Processing of Refugee Claims","year":2023,"lang":"en","type":"article","venue":"","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Computer science; Natural language processing; Inference; Artificial intelligence; Task (project management); Named-entity recognition; Set (abstract data type); Process (computing); Annotation; Refugee; Legal case; Comprehension; Domain (mathematical analysis); Information extraction; Data science; Information retrieval; Machine learning; Law; Political science","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0009491768,0.001218788,0.0007613962,0.007538442,0.00195422,0.001899544,0.002245088,0.002427145,0.0119596],"category_scores_gemma":[0.005413761,0.000336763,0.0007893269,0.007111261,0.0009451128,0.00184259,0.002427391,0.001451441,0.009655947],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00430896,"about_ca_system_score_gemma":0.007666872,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.2156659,"about_ca_topic_score_gemma":0.3706006,"domain_scores_codex":[0.9986737,0.0001862775,0.000188464,0.0002861887,0.0004817284,0.0001837104],"domain_scores_gemma":[0.9964169,0.0009070821,0.0003462906,0.0006591854,0.001337091,0.0003333788],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003000345,0.0002220638,0.009417361,0.001418828,0.00006781767,0.001040819,0.0008001588,0.001810064,0.002613657,0.003565009,0.9473377,0.0314066],"study_design_scores_gemma":[0.0001877493,0.00005850459,0.03060893,0.0004485921,0.00003851452,0.0005993184,0.001495817,0.004603603,0.003130976,0.001902978,0.956845,0.00007991148],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.01763348,0.0008890672,0.001235568,0.0005820192,0.0001166545,0.0002771383,0.9711046,0.001690861,0.006470631],"genre_scores_gemma":[0.008706535,0.0002321173,0.0032076,0.00008217173,0.00001873946,0.0002247579,0.9853241,0.0000794571,0.002124456],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.2156659,"threshold_uncertainty_score":0.4288213,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08610100527460676,"score_gpt":0.457051905237411,"score_spread":0.3709508999628043,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}