{"id":1134064,"url":"https://alion.io/job/icertis-lead-software-engineer-cloud-site-reliability-sre","title":"Lead Software Engineer, Cloud Site Reliability (SRE)","company":{"id":58613,"name":"Icertis","domain":"icertis.com","url":"https://alion.io/company/icertis","size_band":"1001-5000","is_staffing_agency":false,"is_intermediary":false,"ats_vendor":"Oracle","truth_index":{"grade":"A","score":95,"open_postings":5,"ghost_share":0,"stale_share":0.2,"repost_share":0,"time_to_fill_p50_days":34,"computed_at":"2026-09-23T05:45:00Z"}},"role":"DevOps","role_family":"DevOps","seniority":"lead","employment_type":null,"work_mode":"on_site","remote_scope":null,"hiring_geo_confidence":"structured","locations":["Pune, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":31000,"max_usd":74000,"period":"year","method":"global_role_cell_scaled_by_country","sample_n":475},"experience_years_min":7,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AIOps","optional":false},{"name":"Anomaly Detection","optional":false},{"name":"AWS","optional":false},{"name":"Azure","optional":false},{"name":"Azure AKS","optional":false},{"name":"Datadog","optional":false},{"name":"Docker","optional":false},{"name":"Helm","optional":false},{"name":"Incident Management","optional":false},{"name":"Kubernetes","optional":false},{"name":"Power Automate","optional":false},{"name":"Power BI","optional":false},{"name":"PowerShell","optional":false},{"name":"Python","optional":false},{"name":"Self-Healing","optional":false},{"name":"ServiceNow","optional":false},{"name":"SLI/SLO/SLA","optional":false},{"name":"Terraform","optional":false}],"status":"live","first_seen_at":"2026-09-23T04:48:07Z","employer_posted_date":"2026-09-23","last_verified_at":"2026-09-23T13:07:15Z","board_verified":true,"closed_at":null,"days_open":0,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":0},"description":"Required Skills:\n7-12 years of experience in CloudOps/ SRE / NOC environments (24x7 operations)\n\nStrong expertise in Azure Infrastructure (VMs, Networking, Storage)\n\nHands-on experience with Azure Kubernetes Service (AKS), Kubernetes, Docker\n\nStrong experience with monitoring and observability tools (Datadog, Azure Monitor) \n\nProven experience in Incident Management / Major Incident Handling, Monthly reporting\n\nExperience with Infrastructure as Code (Terraform, ARM templates, Helm)\n\nScripting skills in PowerShell, Python, or Bash\n\nExperience with ServiceNow (Incident, Problem, Change modules and dashboards)\n\nGood understanding of distributed systems and cloud-native architecture\n\nExcellent communication, leadership, and problem-solving skills\n\n Role Responsibilities:\nLead 24x7 NOC operations with mandatory rotational shifts ensuring system availability and SLA adherence\n\nAct as Major Incident Manager (P1/P2 incidents), driving triage, war room coordination, and stakeholder communication\n\nImplement and enhance observability practices across logs, metrics, and traces\n\nWork with tools like Datadog and Azure Monitor for monitoring and alerting\n\nDrive proactive monitoring, alert tuning, anomaly detection, and AIOps initiatives\n\nManage Azure infrastructure and AKS clusters, including troubleshooting, scaling, and performance tuning\n\nBuild automation and self-healing workflows using Terraform, ARM, Helm, Power Automate, and scripting\n\nCollaborate with engineering teams to improve reliability, deployment pipelines, and cloud-native architecture\n\nDevelop dashboards and reports using Power BI and ServiceNow\n\nHandle Monthly Business reviews and leadership reporting\n\nMentor team members and drive process standardization and operational excellence\n\n Preferred Certification:\nExperience in multi-cloud environments (Azure/AWS)\n\nExposure to AIOps / predictive monitoring / self-healing systems\n\nAzure / Datadog / Kubernetes certifications","description_format":"text","description_chars":1953,"description_truncated":false,"requirements":{"experience_years_min":7,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Document Management","Business Process Automation (BPA)","Legal Services"],"lifecycle":[{"event":"open","at":"2026-09-23T06:19:09Z"}],"liveness":{"score":86,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.86,"p_room":1,"age_days":0,"expected_fill_days":34,"reasons":["conf:0","win:early"],"computed_at":"2026-09-23T13:09:54Z"},"pay":null,"html_url":"https://alion.io/job/icertis-lead-software-engineer-cloud-site-reliability-sre","json_url":"https://alion.io/job/icertis-lead-software-engineer-cloud-site-reliability-sre.json","meta":{"generated_at":"2026-09-23T13:09:54Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers"}}