{"schemaVersion":"jobsearcher.job.v1","id":"c36100bef5b87797f9e3bd64","url":"https://jobsearcher.com/jobs/c36100bef5b87797f9e3bd64","canonicalUrl":"https://jobsearcher.com/jobs/c36100bef5b87797f9e3bd64","title":"Inference Performance Engineer GPU Kernels & Systems","description":"Acceler8 Talent is seeking a Member of Technical Staff, Inference Performance to accelerate AI model inference across heterogeneous hardware. You will drive optimizations in latency, throughput, and memory usage, and push the serving stack with batching, caching, and quantization techniques.\nThe role involves developing CUDA/HIP/Triton kernels, deploying on accelerator clusters, and benchmarking models under real production workloads to improve efficiency and cost effectiveness.\n\n#J-18808-Ljbffr","company":"Acceler8 Talent","rawCompany":"acceler8 talent","city":"Millbrae","state":"CA","isRemote":false,"isActive":true,"createdAt":"2026-09-15T04:36:26.778Z","occupations":[{"code":"15-1299.08","title":"Computer Systems Engineers/Architects","slug":"computer-systems-engineers-architects"},{"code":"15-1221.00","title":"Computer and Information Research Scientists","slug":"computer-and-information-research-scientists"},{"code":"17-2061.00","title":"Computer Hardware Engineers","slug":"computer-hardware-engineers"}],"industries":[{"code":"518210","title":"Computing Infrastructure Providers, Data Processing, Web Hosting, and Related Services","slug":"computing-infrastructure-providers-data-processing-web-hosting-and-related-services"},{"code":"334111","title":"Electronic Computer Manufacturing","slug":"electronic-computer-manufacturing"},{"code":"541512","title":"Computer Systems Design Services","slug":"computer-systems-design-services"}],"jobPosting":{"@context":"https://schema.org","@type":"JobPosting","title":"Inference Performance Engineer GPU Kernels & Systems","description":"Acceler8 Talent is seeking a Member of Technical Staff, Inference Performance to accelerate AI model inference across heterogeneous hardware. You will drive optimizations in latency, throughput, and memory usage, and push the serving stack with batching, caching, and quantization techniques.\nThe role involves developing CUDA/HIP/Triton kernels, deploying on accelerator clusters, and benchmarking models under real production workloads to improve efficiency and cost effectiveness.\n\n#J-18808-Ljbffr","datePosted":"2026-09-15T04:36:26.778Z","dateModified":"2026-09-15T04:36:26.778Z","hiringOrganization":{"@type":"Organization","name":"Acceler8 Talent","sameAs":"https://jobsearcher.com"},"jobLocation":{"@type":"Place","address":{"@type":"PostalAddress","addressLocality":"Millbrae","addressRegion":"CA","addressCountry":"US"}},"identifier":{"@type":"PropertyValue","name":"JobSearcher","value":"c36100bef5b87797f9e3bd64"},"url":"https://jobsearcher.com/jobs/c36100bef5b87797f9e3bd64"}}