{"id":"W4322764463","doi":"10.1038/s41597-023-02015-3","title":"The Three Terms Task - an open benchmark to compare human and artificial semantic representations","year":2023,"lang":"en","type":"article","venue":"Scientific Data","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Institut Universitaire de Gériatrie de Montréal","funders":"","keywords":"Computer science; Semantic memory; Natural language processing; Artificial intelligence; Semantic similarity; Task (project management); Benchmark (surveying); Word (group theory); Similarity (geometry); Associative property; Noun; Representation (politics); Cognition; Linguistics; Psychology; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003426865,0.001581636,0.0009776242,0.003113632,0.001634864,0.002119375,0.002224751,0.002698108,0.008412775],"category_scores_gemma":[0.02409898,0.0002867641,0.001589213,0.002847889,0.001257949,0.004622595,0.003543478,0.002087227,0.007981929],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00101422,"about_ca_system_score_gemma":0.001296056,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003304636,"about_ca_topic_score_gemma":0.005738501,"domain_scores_codex":[0.9955186,0.001583774,0.0007601236,0.0009095693,0.001048245,0.0001796503],"domain_scores_gemma":[0.98647,0.006318623,0.001236577,0.003309255,0.00186306,0.0008024315],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.004965624,0.002709758,0.07260361,0.007860576,0.001050933,0.001407053,0.005018494,0.01163137,0.02751058,0.02501245,0.4640473,0.3761822],"study_design_scores_gemma":[0.001476218,0.002583086,0.1622446,0.00113185,0.0004218051,0.005091538,0.006430395,0.07181837,0.02435546,0.1175997,0.6062579,0.0005891284],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5001171,0.005151339,0.07038879,0.003152862,0.001816958,0.003209817,0.3336952,0.006973332,0.0754946],"genre_scores_gemma":[0.3820682,0.000840344,0.08740503,0.001314524,0.0004253073,0.005145132,0.5089397,0.001363669,0.01249808],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.008412775,"threshold_uncertainty_score":0.02814353,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1460814197492021,"score_gpt":0.4256313126982627,"score_spread":0.2795498929490606,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}