{"schemaVersion":"jobsearcher.job.v1","id":"88165972ddf3fdae45d25e5f","url":"https://jobsearcher.com/jobs/88165972ddf3fdae45d25e5f","canonicalUrl":"https://jobsearcher.com/jobs/88165972ddf3fdae45d25e5f","title":"Senior GPU Kernel Engineer: Inference Throughput","description":"CoreWeave is seeking a Senior Engineer for its Benchmarking & Performance team to own kernel authoring and optimization for LLM inference. You will write and tune CUDA kernels, optimize tensor cores, and push end-to-end latency down while maintaining accuracy.\nYou will lead benchmarking workflows (MLPerf), mentor engineers, and collaborate with cross-functional partners. Experience with CUDA, C++, Python, and GPU architectures is required; familiarity with vLLM, TensorRT-LLM, llm-d, SGLang is a\n\n#J-18808-Ljbffr","company":"CoreWeave","rawCompany":"coreweave","city":"Seattle","state":"WA","isRemote":false,"isActive":false,"createdAt":"2026-07-24T03:19:19.147Z","occupations":[{"code":"15-1252.00","title":"Software Developers","slug":"software-developers"},{"code":"15-1299.08","title":"Computer Systems Engineers/Architects","slug":"computer-systems-engineers-architects"},{"code":"15-1221.00","title":"Computer and Information Research Scientists","slug":"computer-and-information-research-scientists"}],"industries":[{"code":"513210","title":"Software Publishers","slug":"software-publishers"},{"code":"541511","title":"Custom Computer Programming Services","slug":"custom-computer-programming-services"},{"code":"518210","title":"Computing Infrastructure Providers, Data Processing, Web Hosting, and Related Services","slug":"computing-infrastructure-providers-data-processing-web-hosting-and-related-services"}],"jobPosting":{"@context":"https://schema.org","@type":"JobPosting","title":"Senior GPU Kernel Engineer: Inference Throughput","description":"CoreWeave is seeking a Senior Engineer for its Benchmarking & Performance team to own kernel authoring and optimization for LLM inference. You will write and tune CUDA kernels, optimize tensor cores, and push end-to-end latency down while maintaining accuracy.\nYou will lead benchmarking workflows (MLPerf), mentor engineers, and collaborate with cross-functional partners. Experience with CUDA, C++, Python, and GPU architectures is required; familiarity with vLLM, TensorRT-LLM, llm-d, SGLang is a\n\n#J-18808-Ljbffr","datePosted":"2026-07-24T03:19:19.147Z","dateModified":"2026-07-24T03:19:19.147Z","hiringOrganization":{"@type":"Organization","name":"CoreWeave","sameAs":"https://jobsearcher.com"},"jobLocation":{"@type":"Place","address":{"@type":"PostalAddress","addressLocality":"Seattle","addressRegion":"WA","addressCountry":"US"}},"identifier":{"@type":"PropertyValue","name":"JobSearcher","value":"88165972ddf3fdae45d25e5f"},"url":"https://jobsearcher.com/jobs/88165972ddf3fdae45d25e5f"}}