{"id":"W4416275364","doi":"10.52783/tangence.19","title":"Benchmarking YOLOv4–YOLOv11 for Autonomous Driving: Small-Object Detection, Adverse Conditions and Confidence Calibration","year":2025,"lang":"","type":"article","venue":"Tangence","topic":"Advanced Neural Network Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Calibration; Inference; Brier score; Benchmarking; Detector; Confidence interval; Benchmark (surveying); Throughput","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002274033,0.001743091,0.0009472977,0.0009487018,0.0005124288,0.001056264,0.002362139,0.001390824,0.003254729],"category_scores_gemma":[0.006506985,0.0005324452,0.0007043416,0.0004119722,0.000612549,0.001313155,0.001724861,0.0012306,0.002280078],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008161079,"about_ca_system_score_gemma":0.001656331,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0171128,"about_ca_topic_score_gemma":0.02225105,"domain_scores_codex":[0.9988406,0.0001904531,0.0000621031,0.000402872,0.0002988725,0.0002050125],"domain_scores_gemma":[0.9987743,0.0003519577,0.00007852952,0.000246351,0.0004580252,0.00009071075],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003294891,0.0008615046,0.03132163,0.0009753621,0.0007380944,0.0002778056,0.0003327363,0.3962074,0.04399049,0.001987012,0.02836307,0.49165],"study_design_scores_gemma":[0.0001331323,0.0008196391,0.01191033,0.00008456076,0.00008662239,0.0001898584,0.0001857934,0.9432033,0.03473427,0.0009733144,0.007597046,0.00008204129],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6834971,0.003786184,0.2663905,0.0004430047,0.001218071,0.0004595022,0.002710945,0.02955075,0.011944],"genre_scores_gemma":[0.8781796,0.0004053721,0.1022758,0.0003253972,0.00005273608,0.0002597281,0.01043167,0.001203917,0.006865693],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0171128,"threshold_uncertainty_score":0.03402638,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01963533826006436,"score_gpt":0.2798121024682078,"score_spread":0.2601767642081435,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}