{"id":1908695,"url":"https://alion.io/job/beam-gpu-cluster-infrastructure-engineer","title":"GPU Cluster Infrastructure Engineer","company":{"id":1035,"name":"Beam","domain":"beam.cloud","url":"https://alion.io/company/beam","size_band":"51-200","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Work at a Startup","truth_index":null},"role":"AI/ML","role_family":"AI/ML","seniority":"middle","employment_type":"contractor","work_mode":"remote","remote_scope":"stated_countries","remote_scope_basis":"board_field","remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["New York, United States"],"countries":["US"],"hiring_countries":["US"],"hiring_countries_total":1,"salary":{"min":10500,"max":18000,"currency":"USD","period":"month","gross":null,"usd_annual":216000},"salary_estimate":null,"experience_years_min":3,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Ansible","optional":false},{"name":"GitHub","optional":false},{"name":"Grafana","optional":false},{"name":"HPC","optional":false},{"name":"InfiniBand","optional":false},{"name":"NCCL","optional":false},{"name":"Prometheus","optional":false},{"name":"Snyk","optional":false}],"status":"live","first_seen_at":"2026-10-05T13:50:29Z","employer_posted_date":"2026-10-05","last_verified_at":"2026-10-11T19:41:33Z","board_verified":true,"closed_at":null,"days_open":6,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":6},"description":"Beam is an ultrafast AI inference platform. We built a serverless runtime that launches GPU-backed containers in less than 1 second and quickly scales out to thousands of GPUs. Developers use our platform to serve apps to millions of users around the globe. We're backed by Y Combinator, Tiger Global, and prominent developer-tool founders, including the founder of Snyk and former CTO of GitHub.\nAbout the Role\nWe're building out our own GPU capacity and we're looking for an experienced contractor to help us stand up high-performance GPU clusters. The work runs from design review through bring-in, and you'll leave behind the operational foundation our team needs to run them.\nReview cluster designs and bills of materials across compute, networking, and storage, and catch gaps before hardware is ordered.\nLead acceptance testing: validate cabling and optics, bring up the InfiniBand fabric, run burn-in, and hold vendors to their deliverables.\nStand up and validate high-performance storage alongside vendor teams.\nBuild the out-of-band management layer and firmware baselines, and secure the management plane for customer-facing environments.\nIntegrate hardware, fabric, and storage telemetry into our observability stack, with alerting and automated health checks.\nWrite runbooks, as-builts, and remote-hands procedures.\nProvide escalation support after go-live and help our team ramp up.\nSkills & Experience\nYou've built and operated NVIDIA HGX or DGX clusters in production at a GPU cloud, HPC center, or AI lab.\nHands-on experience with InfiniBand: subnet management and UFM, fabric bring-up, and diagnosing degraded links and optics. NDR or newer.\nGPU node bring-up and burn-in: firmware, BMC/Redfish, DCGM, NCCL testing, PXE and imaging, and XID error triage.\nParallel storage experience: WEKA, VAST, GPFS, Lustre, or similar.\nEqually effective on the data center floor and remotely, including directing colo remote hands.\nYou troubleshoot methodically across hardware, fabric, and software, document as you go, and communicate clearly with technical and non-technical people.\nBonus: recent-generation NVIDIA platforms, bare-metal cloud operations, Ansible or similar automation, Prometheus/Grafana, NVIDIA certifications.\nBenefits\nCompetitive salary and meaningful equity\nJoin a fast-growing pre-series A company at the ground floor\nHealth, dental, and vision benefits with 90% coverage for you and 50% for dependents\nOpportunities to participate in events across the cloud native community\nFitness stipend, learning budget, and much, much more","description_format":"text","description_chars":2558,"description_truncated":false,"requirements":{"experience_years_min":3,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":["Equity"],"hiring_locations":[{"name":"United States","iso":"US","kind":"country"}],"hiring_excludes":[],"relocation_offered":false,"industries":["Cloud Platforms (IaaS & PaaS)","AI Compute & Inference"],"lifecycle":[{"event":"open","at":"2026-10-05T13:50:29Z"}],"visa":[],"liveness":{"score":90,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.903,"p_room":1,"age_days":4,"expected_fill_days":41,"reasons":["conf:3","velocity","win:early"],"computed_at":"2026-10-10T05:45:15Z"},"pay":{"stated_usd_annual":216000,"is_top_pay":true},"html_url":"https://alion.io/job/beam-gpu-cluster-infrastructure-engineer","json_url":"https://alion.io/job/beam-gpu-cluster-infrastructure-engineer.json","meta":{"generated_at":"2026-10-11T19:54:12Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","about":"Alion is a live layer of people, companies and AI agents: who they are, whether they are real and active right now, what they do and how to work with them, readable by people and by agents and paid per call.","catalog":"https://alion.io/catalog.json","usage":{"tier":"crawler_verified","counted_by":"address","units_charged":1,"used_today":7230,"day_limit":null,"remaining_today":null,"minute_limit":300,"resets_at":"2026-10-12T00:00:00Z"}},"offers":[{"id":"company.slices","title":"One company in depth, by slice","status":"live","price":{"credits":0.02,"usd":0.002,"plus_per_slice":{"credits":0.05,"usd":0.005}},"unit":"per company, plus each slice with data","note":"the employer in depth","call":{"mcp_tool":"get_company","arguments":{"id":1035},"rest":"https://alion.io/mcp/rest/get_company?id=1035"},"human":"https://alion.io/catalog?offer=company.slices&for=job%2Fbeam-gpu-cluster-infrastructure-engineer"},{"id":"market.stats","title":"A market slice: pay, demand and time to fill","status":"live","price":{"credits":1,"usd":0.1},"unit":"per slice","note":"pay, demand and time to fill for this role and place","call":{"mcp_tool":"market_stats"},"human":"https://alion.io/catalog?offer=market.stats&for=job%2Fbeam-gpu-cluster-infrastructure-engineer"},{"id":"job.search","title":"Open jobs by role, technology, place, pay and visa","status":"live","price":{"credits":0.02,"usd":0.002},"unit":"per posting in a list","note":"similar open postings","call":{"mcp_tool":"search_jobs"},"human":"https://alion.io/catalog?offer=job.search&for=job%2Fbeam-gpu-cluster-infrastructure-engineer"},{"id":"company.verify","title":"Is this company real and active right now","status":"pilot","price":null,"unit":"per company","request":{"url":"https://alion.io/catalog/request","method":"POST","body":"{\"offer\": \"company.verify\", \"for\": \"job/beam-gpu-cluster-infrastructure-engineer\", \"note\": \"what you need it for\"}"},"human":"https://alion.io/catalog?offer=company.verify&for=job%2Fbeam-gpu-cluster-infrastructure-engineer"}]}