{"id":1640098,"url":"https://alion.io/job/gamma-site-reliability-engineer","title":"Site Reliability Engineer","company":{"id":171803,"name":"Gamma","domain":"gamma.app","url":"https://alion.io/company/gamma-2","size_band":"201-500","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Ashby","truth_index":{"grade":"D","score":54,"open_postings":25,"ghost_share":0.76,"stale_share":0,"repost_share":0,"time_to_fill_p50_days":null,"computed_at":"2026-10-04T05:45:00Z"}},"role":"DevOps","role_family":"DevOps","seniority":"senior","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["San Francisco, United States"],"countries":["US"],"hiring_countries":[],"hiring_countries_total":0,"salary":{"min":230000,"max":310000,"currency":"USD","period":"year","gross":null,"usd_annual":310000},"salary_estimate":null,"experience_years_min":5,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Apache Kafka","optional":false},{"name":"AWS","optional":false},{"name":"Chaos Engineering","optional":false},{"name":"CloudFormation","optional":false},{"name":"Docker","optional":false},{"name":"Go","optional":false},{"name":"Incident Management","optional":false},{"name":"Kubernetes","optional":false},{"name":"Node JS","optional":false},{"name":"Python","optional":false},{"name":"Service Mesh","optional":false},{"name":"SLI/SLO/SLA","optional":false},{"name":"SRE","optional":false},{"name":"Terraform","optional":false},{"name":"TypeScript","optional":false},{"name":"ISO 27001","optional":true},{"name":"JavaScript","optional":true},{"name":"SOC 2","optional":true}],"status":"live","first_seen_at":"2025-11-13T16:53:07Z","employer_posted_date":"2025-11-13","last_verified_at":"2026-10-05T00:36:12Z","board_verified":true,"closed_at":null,"days_open":325,"trust":{"level":"ghost","repost_count":0,"flags":["stale","company_stale"],"days_open":324},"description":"About the role\nGamma's infrastructure needs to be rock-solid for millions of daily users while enabling our engineering teams to ship fast. You'll own the operational health of our full backend platform, building automation and tooling that improves reliability and partnering with engineering to design systems that are observable, resilient, and easy to operate. Your work directly impacts every Gamma user's experience.\nThis is a high-impact role where you'll balance reliability with velocity, knowing when to move fast and when to prioritize stability. You'll lead incident response, drive systemic improvements, and help shape how Gamma scales to serve its next 100 million users.\nOur team has a strong in-office culture and works in person 4-5 days per week in San Francisco. We love working together to stay creative and connected, with flexibility to work from home when focus matters most.\nWhat you'll do\nOwn the reliability, availability, and performance of Gamma's production systems across our AWS infrastructure\n\nBuild observability infrastructure from the ground up: metrics, logging, tracing, and alerting that give the team genuine visibility into system health before users feel the impact\n\nDesign and ship automation that reduces toil, makes deployments safer, and gets us back on our feet faster when things go wrong\n\nLead incident response and blameless post-mortems, then follow through on the systemic fixes that keep the same issues from coming back\n\nPartner with engineering teams on architecture reviews, SLO and SLI design, and reliability best practices that scale with the product\n\nManage and optimize our compute, networking, databases, and managed services\n\nWhat you'll bring\n5+ years in site reliability engineering, DevOps, or systems engineering with deep, hands-on AWS expertise\n\nStrong programming skills in Python, Go, or TypeScript/Node.js, applied to building real tools and automation\n\nSolid experience with infrastructure-as-code (Terraform, CloudFormation) and end-to-end observability solutions\n\nTrack record of making systems meaningfully more reliable through automation, smarter monitoring, and architectural improvements\n\nDeep understanding of networking, distributed systems, containerization (Docker, Kubernetes), and database performance at scale\n\nSharp incident management instincts and the debugging skills to navigate complex production failures\n\nExperience scaling SaaS products to millions of users, or background with Kafka, chaos engineering, or service mesh technologies (Nice to have)\n\nAWS certifications, or experience with security and compliance frameworks like SOC 2 or ISO 27001 (Nice to have)\n\nCompensation range:\nThe base salary for this full-time position, which spans multiple internal levels depending on qualifications, ranges between $230K - $310K plus benefits & equity.\nFinal offer amounts are determined by multiple factors, including but not limited to experience and expertise in the requirements listed above.\nIf you're interested in this role but you don't meet every requirement, we encourage you to apply anyway! We're always excited about meeting great people.","description_format":"text","description_chars":3141,"description_truncated":false,"requirements":{"experience_years_min":5,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":["Equity"],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Artificial Intelligence","Design & Creative","Branding"],"lifecycle":[{"event":"open","at":"2026-10-01T21:56:35Z"}],"visa":[{"country":"US","licensed_sponsor":true,"evidence":"H-1B filings in 12 months: 1","filings_12m":1,"filings_prev_12m":1,"green_card_filings_12m":0,"median_offered_wage_usd":77500,"route":null,"cap_exempt":false,"checked_at":"2026-10-03T21:08:04+00:00","sources":["US Department of Labor: LCA disclosure data (H-1B, H-1B1, E-3)"],"filings_for_role_12m":0}],"liveness":{"score":7,"band":"cold","label":"Long shot","p_open":1,"p_active":0.255,"p_room":0.28,"age_days":324,"expected_fill_days":46,"reasons":["conf:1","stale_co","ghost","win:tail","crowd:"],"computed_at":"2026-10-04T05:45:00Z"},"pay":{"stated_usd_annual":310000,"is_top_pay":true},"html_url":"https://alion.io/job/gamma-site-reliability-engineer","json_url":"https://alion.io/job/gamma-site-reliability-engineer.json","meta":{"generated_at":"2026-10-05T01:16:26Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1639,"day_limit":5000,"remaining_today":3361,"minute_limit":60,"resets_at":"2026-10-06T00:00:00Z"}}}