{"id":"W4405043127","doi":"10.3233/faia241243","title":"Robots in the Middle: Evaluating LLMs in Dispute Resolution","year":2024,"lang":"en","type":"book-chapter","venue":"Frontiers in artificial intelligence and applications","topic":"Law, AI, and Intellectual Property","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Mila - Quebec Artificial Intelligence Institute; Université de Montréal; Research Unit on Children's Psychosocial Maladjustment","funders":"","keywords":"Robot; Resolution (logic); Dispute resolution; Political science; Computer security; Computer science; Artificial intelligence; Law","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01679475,0.001617819,0.000800013,0.002729533,0.001314103,0.004439513,0.003400012,0.003440323,0.006931374],"category_scores_gemma":[0.06658703,0.0003929553,0.00072486,0.001551252,0.001733426,0.006681734,0.004720031,0.002371244,0.003838616],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00219009,"about_ca_system_score_gemma":0.001902958,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003564771,"about_ca_topic_score_gemma":0.00543155,"domain_scores_codex":[0.9766514,0.01550774,0.001141094,0.002352888,0.00395266,0.0003942947],"domain_scores_gemma":[0.9307647,0.05680364,0.002654966,0.005118975,0.00320193,0.00145577],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.005085397,0.003894573,0.03682572,0.005833376,0.000738916,0.0006341772,0.01414054,0.07314512,0.02120353,0.01532014,0.05211543,0.7710631],"study_design_scores_gemma":[0.0008715788,0.004433809,0.02928287,0.001299624,0.0003911728,0.0007999926,0.01406971,0.7463728,0.03711529,0.03753982,0.1274018,0.0004214837],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7582553,0.00746993,0.1539643,0.003123049,0.0007924979,0.002337581,0.005021722,0.0194563,0.04957928],"genre_scores_gemma":[0.7861196,0.000777901,0.1952791,0.0008614399,0.0001339131,0.0009413895,0.008630781,0.0005121828,0.006743626],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01679475,"threshold_uncertainty_score":0.08882022,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1283407056102637,"score_gpt":0.3060759424114476,"score_spread":0.177735236801184,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}