{"schemaVersion":"jobsearcher.job.v1","id":"ae50cb01c06e568a1306f3dd","url":"https://jobsearcher.com/jobs/ae50cb01c06e568a1306f3dd","canonicalUrl":"https://jobsearcher.com/jobs/ae50cb01c06e568a1306f3dd","title":"LLM Inference & Optimization Engineer","description":"Together AI is building scalable AI inference infrastructure to support large language and vision models. You will design and optimize distributed inference engines, focus on low latency, high throughput, and co-design with hardware teams to accelerate GPU/accelerator performance.\nThe role emphasizes research-driven development, collaboration across teams, and delivering end-to-end model serving pipelines in a fast-paced startup environment in San Francisco.\n\n#J-18808-Ljbffr","company":"Together","rawCompany":"together","city":"Millbrae","state":"CA","isRemote":false,"isActive":true,"createdAt":"2026-08-18T04:00:16.116Z","occupations":[{"code":"15-1299.08","title":"Computer Systems Engineers/Architects","slug":"computer-systems-engineers-architects"},{"code":"15-1252.00","title":"Software Developers","slug":"software-developers"},{"code":"15-1221.00","title":"Computer and Information Research Scientists","slug":"computer-and-information-research-scientists"}],"industries":[{"code":"513210","title":"Software Publishers","slug":"software-publishers"},{"code":"541511","title":"Custom Computer Programming Services","slug":"custom-computer-programming-services"},{"code":"518210","title":"Computing Infrastructure Providers, Data Processing, Web Hosting, and Related Services","slug":"computing-infrastructure-providers-data-processing-web-hosting-and-related-services"}],"jobPosting":{"@context":"https://schema.org","@type":"JobPosting","title":"LLM Inference & Optimization Engineer","description":"Together AI is building scalable AI inference infrastructure to support large language and vision models. You will design and optimize distributed inference engines, focus on low latency, high throughput, and co-design with hardware teams to accelerate GPU/accelerator performance.\nThe role emphasizes research-driven development, collaboration across teams, and delivering end-to-end model serving pipelines in a fast-paced startup environment in San Francisco.\n\n#J-18808-Ljbffr","datePosted":"2026-08-18T04:00:16.116Z","dateModified":"2026-08-18T04:00:16.116Z","hiringOrganization":{"@type":"Organization","name":"Together","sameAs":"https://jobsearcher.com"},"jobLocation":{"@type":"Place","address":{"@type":"PostalAddress","addressLocality":"Millbrae","addressRegion":"CA","addressCountry":"US"}},"identifier":{"@type":"PropertyValue","name":"JobSearcher","value":"ae50cb01c06e568a1306f3dd"},"url":"https://jobsearcher.com/jobs/ae50cb01c06e568a1306f3dd"}}