{"id":"W4225383538","doi":"10.32473/flairs.v35i.130643","title":"Learning to Rank with BERT for Argument Quality Evaluation","year":2022,"lang":"en","type":"article","venue":"Proceedings of the ... International Florida Artificial Intelligence Research Society Conference","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Polytechnique Montréal","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Argument (complex analysis); Leverage (statistics); Ranking (information retrieval); Rank (graph theory); Computer science; Learning to rank; Pairwise comparison; Artificial intelligence; Quality (philosophy); Machine learning; Representation (politics); Task (project management); Mathematics; Epistemology; Political science; Engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01331382,0.004083859,0.002532512,0.00779169,0.001296391,0.004887446,0.003469139,0.005560974,0.01239752],"category_scores_gemma":[0.04980415,0.0006579913,0.001715988,0.003708763,0.001570488,0.00578361,0.002402847,0.004842697,0.009445498],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002596239,"about_ca_system_score_gemma":0.002499521,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003933964,"about_ca_topic_score_gemma":0.01022052,"domain_scores_codex":[0.9878461,0.005860652,0.0008078465,0.001401622,0.003409928,0.000673904],"domain_scores_gemma":[0.967267,0.02151415,0.00227224,0.004156947,0.003816488,0.0009731176],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0015995,0.0008104374,0.01171212,0.001701674,0.0004820791,0.0002857404,0.0003485433,0.1813127,0.004878915,0.02226087,0.1159202,0.6586873],"study_design_scores_gemma":[0.0001729476,0.0003827046,0.001926127,0.0001572175,0.0000650724,0.0001868311,0.0001399243,0.9600456,0.004721313,0.01993992,0.01218782,0.00007443786],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1049751,0.01549153,0.7627727,0.003299124,0.001792322,0.001230237,0.01260107,0.06309288,0.03474503],"genre_scores_gemma":[0.5383179,0.00145076,0.4177915,0.001014295,0.0008840281,0.0007173758,0.02352823,0.002058693,0.01423722],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01331382,"threshold_uncertainty_score":0.07041109,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1733581635769031,"score_gpt":0.4124483342358819,"score_spread":0.2390901706589788,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}