{"id":1548833,"url":"https://alion.io/job/johnson-controls-site-reliability-technical-operations-manager","title":"Site Reliability Technical Operations Manager","company":{"id":448321,"name":"Johnson Controls","domain":"johnsoncontrols.com","url":"https://alion.io/company/johnsoncontrols-3","size_band":"5000+","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Workday","truth_index":{"grade":"A","score":88,"open_postings":712,"ghost_share":0.001,"stale_share":0.483,"repost_share":0.008,"time_to_fill_p50_days":24,"computed_at":"2026-10-01T05:45:00Z"}},"role":"Management","role_family":"Management","seniority":"staff","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Pune, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":20000,"max_usd":42000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":8},"experience_years_min":10,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AWS","optional":false},{"name":"Azure","optional":false},{"name":"Confluence","optional":false},{"name":"Datadog","optional":false},{"name":"GCP","optional":false},{"name":"Grafana","optional":false},{"name":"Incident Management","optional":false},{"name":"ITIL","optional":false},{"name":"Jira","optional":false},{"name":"Kubernetes","optional":false},{"name":"ServiceNow","optional":false},{"name":"SLI/SLO/SLA","optional":false}],"status":"live","first_seen_at":"2026-09-30T23:02:01Z","employer_posted_date":"2026-09-30","last_verified_at":"2026-10-01T09:25:34Z","board_verified":true,"closed_at":null,"days_open":0,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":0},"description":"Summary:\nThe Site Reliability Engineering team at Johnson Controls is seeking a Technical Reliability & Support Manager to lead cloud product reliability, operational support, and production stability across global cloud applications and platforms. This role will be responsible for managing day-to-day technical operations, L2/L3 support coordination, incident response, service reliability governance, and continuous improvement for cloud products hosted across platforms such as Azure (primarily), Google Cloud, Ali Cloud, and other cloud environments.\nThe role will partner closely with Engineering, SRE, Security, Observability, and external support partners to ensure production issues are resolved quickly, recurring problems are eliminated, support processes are standardized, and operational risks are proactively identified and addressed.\nPrimary Duties:\nLead technical operations and support management for cloud products across global environments\nManage day-to-day L2/L3 support activities, incident response, escalations, and production issue resolution\nOwn service reliability governance for assigned cloud products, including availability, incident trends, MTTR, recurring issues, and operational risks\nPartner with Engineering, SRE, Security, Observability and Platform teams to improve service stability and operational readiness\nDrive incident management, problem management, RCA/PCA reviews, and corrective action tracking for production issues\nEnsure timely communication during incidents, including stakeholder updates, executive summaries, customer-impact statements, and resolution updates\nEstablish and manage support processes aligned with ITIL practices, including incident, problem, change, service request, and escalation management\nDefine, track, and report operational KPIs such as availability, MTTR, incident volume, severity trends, backlog, service requests, alert noise, and SLA performance\nIdentify recurring operational pain points and drive permanent fixes through engineering backlog, automation, monitoring improvements, and process enhancements\nCollaborate with Observability teams to ensure critical application, infrastructure, database, and integration components are properly monitored and alerted\nSupport implementation and adoption of SLIs, SLOs, SLAs, error budgets, and reliability / supportability scorecards for cloud products\nEnsure production readiness for new product releases, migrations, infrastructure changes, and platform transformations\nLead operational reviews with internal teams, vendors, and external support partners to ensure accountability and continuous improvement\nDrive automation of manual support tasks, ticket workflows, reporting, and operational runbooks to improve efficiency and consistency\nEnsure support activities, incidents, changes, and action items are properly documented and tracked in tools such as Jira, ServiceNow, Confluence, or equivalent platforms\nManage support handoffs across global teams and ensure clear ownership, escalation paths, and communication protocols\nProvide technical leadership during high-severity incidents, major outages, migrations, and critical customer-impacting events\nEnsure compliance with security, audit, operational, and IT governance standards for cloud product support\nMaintain operational documentation, SOPs, runbooks, escalation matrices, support models, and knowledge base articles\nMentor support engineers and technical teams on reliability practices, incident handling, RCA quality, and operational excellence\nQualifications:\n10+ years of experience in technical operations, production support, SRE, cloud support, or reliability engineering.\nStrong experience managing cloud-hosted applications and support operations.\nGood knowledge of Azure (preferred), with exposure to AWS, GCP, or Ali Cloud.\nExperience with incident management, problem management, change management, and operational governance.\nFamiliarity with cloud-native technologies, microservices, APIs, databases, containers, and Kubernetes.\nUnderstanding of observability tools such as Grafana, Datadog, ELK, Logz.io, Azure Monitor, or similar platforms.\nKnowledge of reliability practices including SLIs, SLOs, SLAs, monitoring, alerting, and automation.\nStrong troubleshooting skills across applications, infrastructure, networking, databases, and cloud services.\nExperience with Jira, Confluence, or similar platforms.\nExcellent leadership, communication, stakeholder management, and vendor coordination skills.\nAbility to lead high-severity incidents and drive operational excellence in a fast-paced environment.\nMandatory Skills:\nCloud Operations / SRE leadership\nL2 production support (good to have L2 Production Support)\nIncident & escalation management\nRCA/PCA and problem management\nAzure cloud and cloud-native platforms\nObservability, monitoring & KPI reporting\nITIL-based support processes\nAutomation and runbook development\nExecutive communication & stakeholder management\nLeadership in high-pressure production environments","description_format":"text","description_chars":5031,"description_truncated":false,"requirements":{"experience_years_min":10,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Energy & Utilities","Government","Energy Efficiency","Smart City"],"lifecycle":[{"event":"open","at":"2026-09-30T23:02:01Z"}],"liveness":{"score":90,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.903,"p_room":1,"age_days":0,"expected_fill_days":24,"reasons":["conf:6","velocity","win:early","comp:brand"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/johnson-controls-site-reliability-technical-operations-manager","json_url":"https://alion.io/job/johnson-controls-site-reliability-technical-operations-manager.json","meta":{"generated_at":"2026-10-01T21:02:37Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":3520,"day_limit":5000,"remaining_today":1480,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}