{"id":"W2618664687","doi":"10.3758/s13428-017-0898-2","title":"Scoring best-worst data in unbalanced many-item designs, with applications to crowdsourcing semantic judgments","year":2017,"lang":"en","type":"article","venue":"Behavior Research Methods","topic":"Multi-Criteria Decision Making","field":"Decision Sciences","cited_by":44,"is_retracted":false,"has_abstract":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Crowdsourcing; Computer science; Set (abstract data type); Rank (graph theory); Scaling; Artificial intelligence; Natural language processing; Scoring rule; Machine learning; Data mining; Quality (philosophy); Information retrieval; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.2190499,0.00407681,0.006329006,0.004685143,0.004283062,0.00647405,0.006368466,0.005599225,0.005490639],"category_scores_gemma":[0.5381021,0.002447154,0.005180639,0.004782486,0.008037642,0.00556168,0.009930083,0.005371113,0.000993141],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002589496,"about_ca_system_score_gemma":0.002870946,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002607078,"about_ca_topic_score_gemma":0.003968242,"domain_scores_codex":[0.7003267,0.2652429,0.006930856,0.0161501,0.01003423,0.001315299],"domain_scores_gemma":[0.2824127,0.6634079,0.01105204,0.03116598,0.009726428,0.00223495],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.02004889,0.004328451,0.05817139,0.004687018,0.006906334,0.0006883458,0.01451839,0.2392298,0.00539957,0.1227869,0.01008939,0.5131454],"study_design_scores_gemma":[0.002066419,0.001919264,0.0105227,0.0005064532,0.001001328,0.0001903494,0.001558121,0.7195421,0.002375032,0.2555314,0.004445189,0.0003415779],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.09411725,0.0006382847,0.8984171,0.0005380815,0.0003381961,0.002150957,0.0005444325,0.0008973744,0.002358431],"genre_scores_gemma":[0.3825931,0.00015375,0.6103179,0.0002956528,0.0001747764,0.004090399,0.001197485,0.0002544248,0.0009225649],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.2190499,"threshold_uncertainty_score":0.9630504,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.8789947781800274,"score_gpt":0.7240679895118148,"score_spread":0.1549267886682125,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}