{"id":1626739,"url":"https://alion.io/job/icaremanager-site-reliability-engineer-sre","title":"Site Reliability Engineer (SRE)","company":{"id":2150709,"name":"iCareManager","domain":"icaremanager.com","url":"https://alion.io/company/icaremanager","size_band":null,"is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Zoho Recruit","truth_index":{"grade":"D","score":54,"open_postings":22,"ghost_share":0.773,"stale_share":0,"repost_share":0,"time_to_fill_p50_days":null,"computed_at":"2026-10-06T05:45:30Z"}},"role":"DevOps","role_family":"DevOps","seniority":"middle","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":[],"countries":[],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":null,"experience_years_min":3,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AWS","optional":false},{"name":"Azure","optional":false},{"name":"CI/CD","optional":false},{"name":"Datadog","optional":false},{"name":"Docker","optional":false},{"name":"GCP","optional":false},{"name":"Grafana","optional":false},{"name":"Incident Management","optional":false},{"name":"Kubernetes","optional":false},{"name":"Linux","optional":false},{"name":"Node JS","optional":false},{"name":"Platform Engineering","optional":false},{"name":"Prometheus","optional":false},{"name":"SOC 2","optional":false},{"name":"JavaScript","optional":true}],"status":"live","first_seen_at":"2026-01-28T00:00:00Z","employer_posted_date":"2026-01-28","last_verified_at":"2026-10-06T23:08:43Z","board_verified":true,"closed_at":null,"days_open":251,"trust":{"level":"ghost","repost_count":0,"flags":["stale","company_stale"],"days_open":251},"description":"Role Summary The Site Reliability Engineer (SRE) at iCareManager (iCM) is responsible for ensuring the reliability, scalability, performance, and availability of production systems. The role blends software engineering and operations, with a strong focus on automation, observability, incident management, and proactive reliability engineering. SREs enable engineering teams to deliver features rapidly without compromising system stability, especially in regulated healthcare environments. Key Responsibilities 1. Reliability Engineering Define, implement, and enforce Service Level Indicators (SLIs), Service Level Objectives (SLOs), and error budgets Design and review resilience patterns (redundancy, failover, graceful degradation) Perform capacity planning, load modeling, and scalability analysis Conduct chaos testing and failure injection to identify system weaknesses Reduce Mean Time to Recovery (MTTR) through architectural improvements and tooling 2. Observability & Monitoring Instrument systems with metrics, logs, and distributed traces Build and maintain dashboards that reflect system health and performance Design alerting strategies that are actionable and minimize alert fatigue Identify leading indicators of failure before customer impact 3. Incident Management & Postmortems Participate in and lead production incident response Coordinate with engineering and infrastructure teams during incidents Lead blameless postmortems and document root cause analysis Track and remediate reliability debt and systemic risksRequirements Required\nQualifications Bachelor’s degree in Computer Science, Engineering, or equivalent experience 3+ years experience in SRE, DevOps, Platform Engineering, or similar roles Strong programming experience in one or more languages (e.g., Dot net and or node JS Hands-on experience with Linux-based systems Experience with cloud platforms (Azure preferred; AWS/GCP acceptable) Solid understanding of networking, distributed systems, and system design Experience with monitoring and observability tools (e.g., Prometheus, Grafana, ELK, datadog)\nPreferred\nQualifications Experience in healthcare or regulated environments Familiarity with containerization and orchestration (Docker, Kubernetes) Experience with CI/CD pipelines and infrastructure as code Understanding of security best practices in production systems Experience supporting SOC2-compliant environments Key Competencies Strong problem-solving and analytical skills Calm and effective during high-pressure incidents Excellent documentation and communication skills Ownership mindset and bias toward automation Collaborative and proactive approach\nBenefits A dynamic and collaborative work environment. Opportunities for professional growth and skill development. Competitive salary and benefits package. The chance to play a key role in revolutionizing the healthcare technology industry.","description_format":"text","description_chars":2898,"description_truncated":false,"requirements":{"experience_years_min":3,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":true},"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Health Care","Internet Service Providers","EHR & Clinical Software"],"lifecycle":[{"event":"open","at":"2026-10-01T21:02:24Z"}],"visa":[],"liveness":{"score":3,"band":"cold","label":"Long shot","p_open":1,"p_active":0.112,"p_room":0.28,"age_days":251,"expected_fill_days":18,"reasons":["conf:1","stale_co","velocity","ghost","win:tail","crowd:"],"computed_at":"2026-10-06T05:45:30Z"},"pay":null,"html_url":"https://alion.io/job/icaremanager-site-reliability-engineer-sre","json_url":"https://alion.io/job/icaremanager-site-reliability-engineer-sre.json","meta":{"generated_at":"2026-10-06T23:23:57Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","about":"Alion is a live layer of people, companies and AI agents: who they are, whether they are real and active right now, what they do and how to work with them, readable by people and by agents and paid per call.","catalog":"https://alion.io/catalog.json","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4767,"day_limit":5000,"remaining_today":233,"minute_limit":60,"resets_at":"2026-10-07T00:00:00Z"}},"offers":[{"id":"company.slices","title":"One company in depth, by slice","status":"live","price":{"credits":0.02,"usd":0.002,"plus_per_slice":{"credits":0.05,"usd":0.005}},"unit":"per company, plus each slice with data","note":"the employer in depth","call":{"mcp_tool":"get_company","arguments":{"id":2150709},"rest":"https://alion.io/mcp/rest/get_company?id=2150709"},"human":"https://alion.io/catalog?offer=company.slices&for=job%2Ficaremanager-site-reliability-engineer-sre"},{"id":"market.stats","title":"A market slice: pay, demand and time to fill","status":"live","price":{"credits":1,"usd":0.1},"unit":"per slice","note":"pay, demand and time to fill for this role and place","call":{"mcp_tool":"market_stats"},"human":"https://alion.io/catalog?offer=market.stats&for=job%2Ficaremanager-site-reliability-engineer-sre"},{"id":"job.search","title":"Open jobs by role, technology, place, pay and visa","status":"live","price":{"credits":0.02,"usd":0.002},"unit":"per posting in a list","note":"similar open postings","call":{"mcp_tool":"search_jobs"},"human":"https://alion.io/catalog?offer=job.search&for=job%2Ficaremanager-site-reliability-engineer-sre"},{"id":"company.verify","title":"Is this company real and active right now","status":"pilot","price":null,"unit":"per company","request":{"url":"https://alion.io/catalog/request","method":"POST","body":"{\"offer\": \"company.verify\", \"for\": \"job/icaremanager-site-reliability-engineer-sre\", \"note\": \"what you need it for\"}"},"human":"https://alion.io/catalog?offer=company.verify&for=job%2Ficaremanager-site-reliability-engineer-sre"}]}