{"id":1431526,"url":"https://alion.io/job/heygen-software-engineer-gpu-performance","title":"Software Engineer, GPU Performance","company":{"id":48080,"name":"HeyGen","domain":"heygen.com","url":"https://alion.io/company/heygen","size_band":"51-200","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Greenhouse","truth_index":{"grade":"B","score":75,"open_postings":18,"ghost_share":0.333,"stale_share":0,"repost_share":0,"time_to_fill_p50_days":165,"computed_at":"2026-10-01T05:45:00Z"}},"role":"AI/ML","role_family":"AI/ML","seniority":null,"employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Los Angeles, United States"],"countries":["US"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":null,"experience_years_min":null,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Machine Learning","optional":false},{"name":"Python","optional":false},{"name":"PyTorch","optional":false},{"name":"PyTorch C++","optional":false},{"name":"C++","optional":true},{"name":"CUDA","optional":true},{"name":"CUDA Toolkit","optional":true},{"name":"Quantization","optional":true},{"name":"Transformers","optional":true},{"name":"Triton","optional":true}],"status":"live","first_seen_at":"2026-09-28T22:27:25Z","employer_posted_date":"2026-09-28","last_verified_at":"2026-10-01T06:10:46Z","board_verified":true,"closed_at":null,"days_open":2,"trust":{"level":"ok","repost_count":0,"flags":["company_stale"],"days_open":2},"description":"About HeyGen\nAt HeyGen, our mission is to make visual storytelling accessible to all. Over the last decade, visual content has become the preferred method of information creation, consumption, and retention. But the ability to create such content, in particular videos, continues to be costly and challenging to scale. Our ambition is to build technology that equips more people with the power to reach, captivate, and inspire audiences.\nLearn more at www.heygen.com. Visit our Mission and Culture doc here. \nPosition Summary\nHeyGen is building AI applications including Avatar IV, Photo Avatar, Interactive Avatar, and Video Translation. We’re looking for a Software Engineer focused on GPU performance to make the systems behind these experiences faster and more efficient.\nYou will work across model execution and inference infrastructure, using profiling and measurement to improve latency, throughput, and GPU cost. This role is a fit for an engineer who enjoys understanding how software uses the hardware beneath it.\nKey Responsibilities\nUse NVIDIA Nsight Systems, Nsight Compute, and PyTorch Profiler to investigate GPU utilization, kernel execution, memory bandwidth, and CPU-GPU data movement.\nIdentify bottlenecks across model execution, preprocessing, and inference serving, then measure the impact of each optimization.\nImprove performance through batching, scheduling, memory management, and better GPU utilization.\nDevelop or integrate high-performance GPU kernels when existing implementations limit performance.\nBuild benchmarks and automated checks that catch performance regressions across representative video workloads.\nCollaborate with AI researchers and infrastructure engineers to bring optimizations into production.\nMeasure the effect of changes on latency, throughput, cost, and output quality.\nQualifications\nExperience optimizing GPU-based AI workloads or high-performance computing systems.\nProficiency in Python and experience with PyTorch or a similar machine learning framework.\nStrong curiosity about GPU hardware, including memory bandwidth, cache behavior, tensor cores, and data movement between CPU and GPU.\nExperience using profiling tools to connect hardware behavior to application-level bottlenecks and validate improvements.\nAbility to turn performance experiments into reliable production changes and communicate tradeoffs clearly.\nPreferred Qualifications\nExperience with CUDA, Triton, or C++ GPU programming.\nExperience optimizing video, image, audio, diffusion, or Transformer models.\nFamiliarity with multi-GPU inference, GPU interconnects, quantization, or large-scale model serving.\nExperience building performance benchmarks or regression testing infrastructure.\nPrior experience in a fast-paced technology environment.\nWhat HeyGen Offers\nCompetitive salary and benefits package.\nDynamic and inclusive work environment.\nOpportunities for professional growth and advancement.\nCollaborative culture that values innovation and creativity.\nAccess to the latest technologies and tools.\nHeyGen is an Equal Opportunity Employer. We celebrate diversity and are committed to creating an inclusive environment for all employees.\nJoin us at HeyGen\nHelp us make AI video faster and more accessible. We’d love to hear from you.","description_format":"text","description_chars":3265,"description_truncated":false,"requirements":{"experience_years_min":null,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Generative Video"],"lifecycle":[{"event":"open","at":"2026-09-29T01:36:17Z"}],"liveness":{"score":90,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.903,"p_room":1,"age_days":2,"expected_fill_days":165,"reasons":["conf:11","velocity","win:early"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/heygen-software-engineer-gpu-performance","json_url":"https://alion.io/job/heygen-software-engineer-gpu-performance.json","meta":{"generated_at":"2026-10-01T11:10:55Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":2649,"day_limit":5000,"remaining_today":2351,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}