{"id":1252736,"url":"https://alion.io/job/openinfer-ai-model-architecture-optimization-engineer-rd","title":"AI Model Architecture Optimization Engineer R&D","company":{"id":35877,"name":"OpenInfer","domain":"openinfer.io","url":"https://alion.io/company/openinfer","size_band":null,"is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Schema","truth_index":null},"role":"AI/ML","role_family":"AI/ML","seniority":null,"employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["San Mateo, United States"],"countries":["US"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":155000,"max_usd":335000,"period":"year","method":"role_country_seniority_unknown","sample_n":2391},"experience_years_min":null,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"CUDA","optional":false},{"name":"CUDA Toolkit","optional":false},{"name":"Python","optional":false},{"name":"PyTorch","optional":false},{"name":"KV Cache","optional":true},{"name":"Tokenization","optional":true}],"status":"live","first_seen_at":"2024-12-24T00:00:00Z","employer_posted_date":"2024-12-24","last_verified_at":"2026-09-26T02:41:38Z","board_verified":true,"closed_at":null,"days_open":642,"trust":{"level":"stale","repost_count":0,"flags":["stale"],"days_open":641},"description":"Position Overview\nWe are looking for an experienced AI Acceleration Engineer who can dive deep into large model (eg. transformer) architectures and blocks such as self/cross/multi-attention, and perform research and development of advanced techniques to accelerate these areas. The ideal candidate will have a deep understanding of large model design, AI acceleration techniques, and will integrate these advancements into the PyTorch stack. Familiarity with Python is essential, and experience with CUDA programming is highly desirable.\nKey Responsibilities\nInnovate on AI model components, such as attention blocks, KV-cache strategies, layer streaming, tokenization, layer norms, and more, to improve AI model performance and scalability.\nOptimize and integrate AI acceleration techniques into the PyTorch stack, enabling efficient use across diverse hardware platforms.\nOwn & drive features end to end to push the limits of large model architecture, ensuring seamless integration with existing frameworks.\nBenchmark and profile AI models to evaluate performance improvements, ensuring optimal execution on target hardware.\nWrite and maintain clean, efficient code in Python, with a focus on integration with PyTorch.\nLeverage CUDA for GPU-based acceleration when necessary, optimizing the attention blocks for maximum performance.\nWork on cross-functional teams to design, implement, and test new features.\nQualifications\nExtensive experience with large AI model architectures, particularly with attention blocks and transformer models.\nProficiency in Python and hands-on experience with the PyTorch framework.\nStrong understanding of AI acceleration techniques and their application in real-world use cases.\nFamiliarity with CUDA for GPU programming is highly desirable.\nDemonstrated ability to optimize complex models for performance across different hardware environments.\nExperience in developing and deploying AI models at scale is a plus.\nWhat You’ll Gain\nOpportunity to work alongside industry experts in AI optimization, high-performance computing, and hardware acceleration.\nHands-on experience with cutting-edge technologies at the intersection of AI and hardware acceleration.\nExposure to open-source development and collaboration with a vibrant community.\nBenefits We Offer:\nAt OpenInfer we offer comprehensive benefits, some include:\nMedical, Dental, and Vision benefits\nFlexible Paid Time Off, 10 days\nParental Leave\n401(k) Plan with company matching\nSnacks and coffee to keep you energized\nThese benefits are further detailed in OpenInfer policies and are subject to change at any time, consistent with the terms of any applicable compensation or benefits plans.","description_format":"text","description_chars":2681,"description_truncated":false,"requirements":{"experience_years_min":null,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":["Parental leave"],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Corporate Events","AI Compute & Inference"],"lifecycle":[{"event":"open","at":"2026-09-25T18:38:11Z"}],"liveness":{"score":4,"band":"cold","label":"Long shot","p_open":1,"p_active":0.126,"p_room":0.28,"age_days":641,"expected_fill_days":21,"reasons":["conf:3","win:tail","crowd:"],"computed_at":"2026-09-26T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/openinfer-ai-model-architecture-optimization-engineer-rd","json_url":"https://alion.io/job/openinfer-ai-model-architecture-optimization-engineer-rd.json","meta":{"generated_at":"2026-09-27T02:26:00Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":2167,"day_limit":5000,"remaining_today":2833,"minute_limit":60,"resets_at":"2026-09-28T00:00:00Z"}}}