{"id":"W6910410365","doi":"10.48448/mrw0-4a11","title":"Re-Invoke: Tool Invocation Rewriting for Zero-Shot Tool Retrieval","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Leverage (statistics); Search engine indexing; Ranking (information retrieval); Bottleneck; Key (lock); Set (abstract data type); Inference; SPARQL; Context (archaeology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002166576,0.003067501,0.002011956,0.004002589,0.001166594,0.002551184,0.005155156,0.002178683,0.007886908],"category_scores_gemma":[0.01041323,0.0006963109,0.002338399,0.002673586,0.001357105,0.004395165,0.004507212,0.002131138,0.00827481],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001110689,"about_ca_system_score_gemma":0.002370716,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009727106,"about_ca_topic_score_gemma":0.02238409,"domain_scores_codex":[0.9957463,0.001059089,0.000343829,0.001106819,0.001401398,0.0003424859],"domain_scores_gemma":[0.9952998,0.001516034,0.0002166475,0.002111695,0.0006144962,0.0002413207],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0009707474,0.0008068282,0.004165514,0.001584065,0.0003627249,0.0007226852,0.000995681,0.02825318,0.0681388,0.01684074,0.140729,0.73643],"study_design_scores_gemma":[0.0003402312,0.0005328045,0.00215849,0.000101745,0.000166193,0.001087805,0.0006143717,0.8287718,0.05542846,0.03103048,0.07955085,0.0002167559],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02943556,0.002319972,0.7351256,0.0004512686,0.0003065004,0.0007273357,0.005542304,0.2166121,0.009479309],"genre_scores_gemma":[0.1918601,0.0006406561,0.7586227,0.0006707845,0.0001504147,0.0004956885,0.02819827,0.008600142,0.01076125],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.009727106,"threshold_uncertainty_score":0.02638435,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06129313179942047,"score_gpt":0.3517099969563268,"score_spread":0.2904168651569063,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}