{"id":1228814,"url":"https://alion.io/job/searce-lead-senior-site-reliability-engineer","title":"Lead | Senior Site Reliability Engineer","company":{"id":180441,"name":"Searce","domain":"searce.com","url":"https://alion.io/company/searce","size_band":null,"is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"DevOps","role_family":"DevOps","seniority":"lead","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Pune, India","Mumbai, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":{"min":600000,"max":2200000,"currency":"INR","period":"year","gross":true,"usd_annual":23060},"salary_estimate":null,"experience_years_min":4,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":".NET","optional":false},{"name":"Amazon CloudWatch","optional":false},{"name":"Amazon EC2","optional":false},{"name":"Amazon ECS","optional":false},{"name":"Amazon EKS","optional":false},{"name":"Ansible","optional":false},{"name":"AWS","optional":false},{"name":"AWS Lambda","optional":false},{"name":"Chef","optional":false},{"name":"Docker","optional":false},{"name":"GCP","optional":false},{"name":"Google GKE","optional":false},{"name":"Incident Management","optional":false},{"name":"JavaScript","optional":false},{"name":"Kubernetes","optional":false},{"name":"Linux","optional":false},{"name":"Puppet","optional":false},{"name":"Python","optional":false},{"name":"Ruby","optional":false},{"name":"Terraform","optional":false},{"name":"Windows","optional":false},{"name":"C#","optional":true}],"status":"live","first_seen_at":"2026-09-22T06:07:07Z","employer_posted_date":null,"last_verified_at":"2026-09-22T06:07:07Z","board_verified":false,"closed_at":null,"days_open":9,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":9},"description":"Lead Cloud Reliability Engineer\n\nJob Responsibilities\n● Lead and manage the Cloud Reliability teams to provide strong Managed Services support to end-customers.\n● Isolate, troubleshoot and resolve issues reported by CMS clients in their cloud environment\n● Drive the communication with the customer providing details about the issue, current steps, next plan of action, ETA\n● Gather client's requirements related to use of specic cloud services and provide assistance in seing them up and resolving issues\n● Create SOPs and knowledge articles for use by the L1 teams to resolve common issues\n● Identify recurring issues, perform root cause analysis and propose/implement preventive actions\n● Follow change management procedure to identify, record and implement changes\n● Plan and deploy OS, security patches in Windows/Linux environment and upgrade k8s clusters\n● Identify the recurring manual activities and contribute to automation\n● Provide technical guidance and educate team members on development and operations. Monitor metrics and develop ways to improve.\n● System troubleshooting and problem-solving across plaorm and application domains. Ability to use a wide variety of open-source technologies and cloud services.\n● Build, maintain, and monitor conguration standards.\n● Ensuring critical system security through using best-in-class cloud security solutions.\n\nQualifications\n● 4-7 years experience in Cloud Infrastructure and Operations domains and IT operational experience preferably in a global enterprise environment.\n● Specialize in one or two cloud deployment platforms: AWS, GCP\n● Hands on experience with AWS/GCP services (EKS, ECS, EC2, VPC, RDS, Lambda, GKE, Compute Engine)\n● Understanding of one or more programming languages (Python, JavaScript, Ruby, Java, .Net)\n● Logging and Monitoring tools (ELK, Stackdriver, CloudWatch)\n● Knowledge on Conguration Management tools such as Ansible, Terraform, Puppet, Chef\n● Experience working with deployment and orchestration technologies (such as Docker, Kubernetes, Mesos)\n● Good analytical, communication, problem solving, and learning skills.\n● Knowledge on programming against cloud plaorms such as Google Cloud Platform and lean development methodologies.\n● Strong service aitude and a commitment to quality.\n● Willingness to work in shifts.\nRead less\n\nSkills\nReliability engineering, DevOps, Google Cloud Platform (GCP), Alerting and Monitoring, Kubernetes, Network Security, Incident management, Observability, IT operations, Infrastructure, L2","description_format":"text","description_chars":2516,"description_truncated":false,"requirements":{"experience_years_min":4,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Artificial Intelligence","Science & Engineering","Engineering Services"],"lifecycle":[{"event":"open","at":"2026-09-25T14:00:00Z"}],"liveness":{"score":75,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.794,"p_room":0.945,"age_days":8,"expected_fill_days":24,"reasons":["seen:8","velocity","win:mid"],"computed_at":"2026-10-01T05:45:00Z"},"pay":{"stated_usd_annual":23060,"is_top_pay":false},"html_url":"https://alion.io/job/searce-lead-senior-site-reliability-engineer","json_url":"https://alion.io/job/searce-lead-senior-site-reliability-engineer.json","meta":{"generated_at":"2026-10-01T21:30:03Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4150,"day_limit":5000,"remaining_today":850,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}