{"id":1701358,"url":"https://alion.io/job/jumpcloud-senior-site-reliability-engineer-india","title":"Senior Site Reliability Engineer - India","company":{"id":2492,"name":"JumpCloud","domain":"jumpcloud.com","url":"https://alion.io/company/jump-cloud","size_band":"201-500","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Lever","truth_index":{"grade":"B","score":80,"open_postings":5,"ghost_share":0,"stale_share":0.8,"repost_share":0,"time_to_fill_p50_days":35,"computed_at":"2026-10-02T05:45:00Z"}},"role":"DevOps","role_family":"DevOps","seniority":"senior","employment_type":"full_time","work_mode":"remote","remote_scope":"stated_countries","remote_scope_basis":"posting_text","remote_working_hours":null,"hiring_geo_confidence":"explicit","locations":["Bengaluru, India"],"countries":["IN"],"hiring_countries":["IN"],"hiring_countries_total":1,"salary":null,"salary_estimate":{"min_usd":18000,"max_usd":43000,"period":"year","method":"role_seniority_country_cell","sample_n":48},"experience_years_min":8,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Amazon EKS","optional":false},{"name":"ArgoCD","optional":false},{"name":"AWS","optional":false},{"name":"Claude Code","optional":false},{"name":"Copilot","optional":false},{"name":"Cursor","optional":false},{"name":"Datadog","optional":false},{"name":"Error Budget","optional":false},{"name":"FinOps","optional":false},{"name":"GCP","optional":false},{"name":"GitOps","optional":false},{"name":"Google GKE","optional":false},{"name":"HAProxy","optional":false},{"name":"IAM","optional":false},{"name":"Incident Management","optional":false},{"name":"Istio","optional":false},{"name":"Kubernetes","optional":false},{"name":"Linkerd","optional":false},{"name":"Nginx","optional":false},{"name":"PagerDuty","optional":false},{"name":"Platform Engineering","optional":false},{"name":"Python","optional":false},{"name":"SLI/SLO/SLA","optional":false},{"name":"Terraform","optional":false},{"name":"cert-manager","optional":true},{"name":"Chaos Engineering","optional":true},{"name":"External Secrets","optional":true}],"status":"live","first_seen_at":"2026-09-29T17:53:11Z","employer_posted_date":"2026-09-29","last_verified_at":"2026-10-02T13:31:01Z","board_verified":true,"closed_at":null,"days_open":3,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":3},"description":"What You’ll Be Doing:\nArchitect, scale, and continuously improve the reliability, availability, and performance of JumpCloud’s multi-region microservices, APIs, and authentication infrastructure (AWS/GCP).\n\nArchitect, build, and maintain Disaster Recovery (DR) process, multi-region failover automation, and business continuity strategies to ensure rapid recovery against strict RTO and RPO objectives.\n\nLead the design and enforcement of SLIs, SLOs, and Error Budget frameworks across multi-disciplinary engineering teams.\n\nDrive end-to-end observability strategy using Datadog, implementing actionable Golden Signals monitoring to drastically reduce MTTD/MTTR and eliminate alert fatigue.\n\nLead on-call escalation, major incident management, and drive strict adherence to 99.99% availability SLAs.\n\nFacilitate blameless post-incident reviews, executing systemic root-cause remediations to prevent recurring failure modes.\n\nArchitect, manage, and scale production Kubernetes (EKS) clusters, implementing advanced GitOps workflows (Argo CD, Kargo) and deployment patterns.\n\nDesign and maintain modular, enterprise-grade Infrastructure-as-Code using Terraform across multi-account, multi-region cloud environments.\n\nDesign, build, and maintain interactive FinOps and cost-optimization dashboards to provide engineering and leadership teams with actionable insights into multi-cloud spend, unit economics, and resource utilization.\n\nEliminate complex operational toil by writing production-grade Python or Go tooling, platform automation, and custom integrations.\n\nChampion AI-assisted software development workflows (Cursor, Claude Code, GitHub Copilot) to accelerate automation, runbook creation, and incident triage across the team.\n\nAuthor operational runbooks, architecture decision records, and mentor mid-level/junior engineers to raise the overall technical bar.\n\nWe’re Looking For:\n8+ years of professional software engineering experience in SRE, DevOps, or Platform Engineering operating 24/7 mission-critical, highly available distributed systems.\n\nBachelor's degree in Computer Science, Software Engineering, or equivalent technical discipline.\n\nStrong Python/Go Capabilities: Advanced software engineering skill set for writing internal SRE platforms, tools, and API integrations.\n\nDeep Kubernetes Expertise: Hands-on experience with production EKS/GKE cluster lifecycles, ingress/egress, networking, RBAC, and GitOps tooling (Argo CD).\n\nAdvanced IaC & AWS/GCP: Deep Terraform proficiency (module architecture, state management refactoring) across complex multi-account AWS environments (IAM, VPCs, Transit Gateway, ALB/NLB, Route53).\n\nFinOps & Cost Optimization Leadership: Demonstrated experience driving cloud cost-efficiency strategies, resource right-sizing, cost-allocation tagging, workload optimization, and building FinOps dashboards to embed financial accountability into engineering workflows.\n\nDisaster Recovery & High Availability: Proven background in designing and testing multi-region Disaster Recovery architectures, automating failover systems, and monitoring recovery health via DR dashboards.\n\nObservability & Reliability Architecture: Track record of defining SLI/SLOs, managing PagerDuty schedules, and optimizing production observability platforms.\n\nExperience designing and operating enterprise service meshes (Istio, Linkerd, or similar) and production ingress/proxy systems (HAProxy, NGINX, or similar).\n\nTechnical Mentorship: Demonstrated ability to lead technical discussions, write architectural design docs/RFCs, and mentor engineering peers.\n\nStrong problem-solving, communication, and collaboration skills with a passion for solving complex distributed systems challenges at scale.\n\nA strong team player who helps us live by our core values: building connections, thinking big, and getting 1% better every day.\n\nPreferred Qualifications:\nBasic understanding of chaos engineering principles or testing resilience in staging/production.\n\nExperience with secrets management architectures (Vault, AWS Secrets Manager, External Secrets Operator, Cert-Manager).\n\nBackground in DevSecOps practices, service meshes (Istio), and automated vulnerability remediation within cloud infrastructure code.\n\nBackground supporting identity services, IAM, enterprise directory platforms, or security-focused SaaS solutions.","description_format":"text","description_chars":4347,"description_truncated":false,"requirements":{"experience_years_min":8,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":false},"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[{"name":"India","iso":"IN","kind":"country"}],"hiring_excludes":[],"relocation_offered":false,"industries":["Identity Management"],"lifecycle":[{"event":"open","at":"2026-10-02T13:31:01Z"}],"liveness":{"score":86,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.86,"p_room":1,"age_days":3,"expected_fill_days":35,"reasons":["conf:14","win:early"],"computed_at":"2026-10-03T04:02:12Z"},"pay":null,"html_url":"https://alion.io/job/jumpcloud-senior-site-reliability-engineer-india","json_url":"https://alion.io/job/jumpcloud-senior-site-reliability-engineer-india.json","meta":{"generated_at":"2026-10-03T04:02:12Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4068,"day_limit":5000,"remaining_today":932,"minute_limit":60,"resets_at":"2026-10-04T00:00:00Z"}}}