{"schemaVersion":"jobsearcher.job.v1","id":"6f45f0b7e02b06de42a27041","url":"https://jobsearcher.com/jobs/6f45f0b7e02b06de42a27041","canonicalUrl":"https://jobsearcher.com/jobs/6f45f0b7e02b06de42a27041","title":"GPU Systems Engineer Distributed Training & Inference","description":"TensorScale AI in San Francisco seeks a hardware‑aware software engineer to optimize GPU systems for training and inference across image, video, and world‑model workloads. You will push performance from kernel code to distributed engines, profiling bottlenecks and implementing practical improvements that scale.\nResponsibilities include CUDA / Triton optimizations, designing efficient distributed inference and training pipelines, and owning communication performance across GPUs and nodes with\n\n#J-18808-Ljbffr","company":"Tensorscale Ai","rawCompany":"tensorscale ai","city":"Millbrae","state":"CA","isRemote":false,"isActive":false,"createdAt":"2026-08-23T03:18:09.597Z","occupations":[{"code":"15-1299.08","title":"Computer Systems Engineers/Architects","slug":"computer-systems-engineers-architects"},{"code":"15-1252.00","title":"Software Developers","slug":"software-developers"},{"code":"15-1221.00","title":"Computer and Information Research Scientists","slug":"computer-and-information-research-scientists"}],"industries":[{"code":"513210","title":"Software Publishers","slug":"software-publishers"},{"code":"541512","title":"Computer Systems Design Services","slug":"computer-systems-design-services"},{"code":"518210","title":"Computing Infrastructure Providers, Data Processing, Web Hosting, and Related Services","slug":"computing-infrastructure-providers-data-processing-web-hosting-and-related-services"}],"jobPosting":{"@context":"https://schema.org","@type":"JobPosting","title":"GPU Systems Engineer Distributed Training & Inference","description":"TensorScale AI in San Francisco seeks a hardware‑aware software engineer to optimize GPU systems for training and inference across image, video, and world‑model workloads. You will push performance from kernel code to distributed engines, profiling bottlenecks and implementing practical improvements that scale.\nResponsibilities include CUDA / Triton optimizations, designing efficient distributed inference and training pipelines, and owning communication performance across GPUs and nodes with\n\n#J-18808-Ljbffr","datePosted":"2026-08-23T03:18:09.597Z","dateModified":"2026-08-23T03:18:09.597Z","hiringOrganization":{"@type":"Organization","name":"Tensorscale Ai","sameAs":"https://jobsearcher.com"},"jobLocation":{"@type":"Place","address":{"@type":"PostalAddress","addressLocality":"Millbrae","addressRegion":"CA","addressCountry":"US"}},"identifier":{"@type":"PropertyValue","name":"JobSearcher","value":"6f45f0b7e02b06de42a27041"},"url":"https://jobsearcher.com/jobs/6f45f0b7e02b06de42a27041"}}