{"schemaVersion":"jobsearcher.job.v1","id":"cdcf2fcfa19362ad13add490","url":"https://jobsearcher.com/jobs/cdcf2fcfa19362ad13add490","canonicalUrl":"https://jobsearcher.com/jobs/cdcf2fcfa19362ad13add490","title":"Machine Learning Inference Engineer","description":"An AI Unicorn startup is hiring a Senior Machine Learning Inference Engineer for a full-time role. You will be responsible for improving efficiency for AI-native infrastructure powered by generative and multimodal models. The ideal candidate has over 3 years of professional experience and a strong understanding of GPU infrastructure, Python, and PyTorch. This is a highly autonomous role with significant ownership across inference systems and model performance in production. This role is hybrid in San Francisco Bay Area and offers full benefits and equity. Experience: Building AI applications at scale from the ground up Strong understanding of GPU infrastructure including Triton, TensorRT, or vLLM frameworks Hands-on experience with Python and PyTorch Building model-serving Microservices Diffusion and Multimodal model experience is a plus Equity $401k matching Medical coverage","company":"Oscar","rawCompany":"oscar","city":"Millbrae","state":"CA","isRemote":false,"isActive":false,"createdAt":"2026-09-10T13:23:05.237Z","occupations":[{"code":"15-1299.08","title":"Computer Systems Engineers/Architects","slug":"computer-systems-engineers-architects"},{"code":"15-2051.00","title":"Data Scientists","slug":"data-scientists"},{"code":"15-1252.00","title":"Software Developers","slug":"software-developers"}],"industries":[{"code":"541511","title":"Custom Computer Programming Services","slug":"custom-computer-programming-services"},{"code":"541512","title":"Computer Systems Design Services","slug":"computer-systems-design-services"},{"code":"513210","title":"Software Publishers","slug":"software-publishers"}],"jobPosting":{"@context":"https://schema.org","@type":"JobPosting","title":"Machine Learning Inference Engineer","description":"An AI Unicorn startup is hiring a Senior Machine Learning Inference Engineer for a full-time role. You will be responsible for improving efficiency for AI-native infrastructure powered by generative and multimodal models. The ideal candidate has over 3 years of professional experience and a strong understanding of GPU infrastructure, Python, and PyTorch. This is a highly autonomous role with significant ownership across inference systems and model performance in production. This role is hybrid in San Francisco Bay Area and offers full benefits and equity. Experience: Building AI applications at scale from the ground up Strong understanding of GPU infrastructure including Triton, TensorRT, or vLLM frameworks Hands-on experience with Python and PyTorch Building model-serving Microservices Diffusion and Multimodal model experience is a plus Equity $401k matching Medical coverage","datePosted":"2026-09-10T13:23:05.237Z","dateModified":"2026-09-10T13:23:05.237Z","hiringOrganization":{"@type":"Organization","name":"Oscar","sameAs":"https://jobsearcher.com"},"jobLocation":{"@type":"Place","address":{"@type":"PostalAddress","addressLocality":"Millbrae","addressRegion":"CA","addressCountry":"US"}},"identifier":{"@type":"PropertyValue","name":"JobSearcher","value":"cdcf2fcfa19362ad13add490"},"url":"https://jobsearcher.com/jobs/cdcf2fcfa19362ad13add490"}}