{"id":1247905,"url":"https://alion.io/job/metarouter-principal-site-reliability-engineer","title":"Principal Site Reliability Engineer","company":{"id":1957787,"name":"MetaRouter","domain":"metarouter.io","url":"https://alion.io/company/metarouter","size_band":"11-50","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Deel","truth_index":{"grade":"B","score":75,"open_postings":3,"ghost_share":0,"stale_share":1,"repost_share":0,"time_to_fill_p50_days":null,"computed_at":"2026-10-01T05:45:00Z"}},"role":"DevOps","role_family":"DevOps","seniority":"lead","employment_type":"full_time","work_mode":"remote","remote_scope":"stated_countries","remote_scope_basis":"inferred_company_offices","remote_working_hours":null,"hiring_geo_confidence":"inferred","locations":[],"countries":[],"hiring_countries":["GB"],"hiring_countries_total":1,"salary":{"min":180000,"max":250000,"currency":"USD","period":"year","gross":null,"usd_annual":250000},"salary_estimate":null,"experience_years_min":10,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"CI/CD","optional":false},{"name":"Linux","optional":false},{"name":"Platform Engineering","optional":false},{"name":"Unix","optional":false}],"status":"live","first_seen_at":"2026-09-16T18:44:49Z","employer_posted_date":"2026-09-16","last_verified_at":"2026-09-30T12:36:31Z","board_verified":true,"closed_at":null,"days_open":14,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":14},"description":"About The Role\nAs a Principal Site Reliability Engineer, you set the reliability strategy for the platform. You will define how we build, deploy, observe, and operate a distributed system that runs both in our own cloud and inside customer-controlled environments - and you will hold the organization to that standard.\nWe run dedicated, isolated environments per customer, which makes repeatability and automation the central engineering problem rather than an afterthought. Depth of judgment about reliability engineering matters far more here than experience with any particular cloud, orchestrator, or observability vendor.\nThis is an individual contributor role with organization-level influence.\nCore Responsibilities\nOwn the reliability architecture of the platform: deployment topology, failure domains, capacity strategy, and the automation that makes environments reproducible.\n\nDefine service level objectives with product and engineering leadership, and drive the work needed to meet them.\n\nSet the standard for observability - dashboards, logs, metrics, tracing, and alerting - so that issues are detected before customers report them.\n\nLead major incidents, run blameless postmortems, and make sure the corrective work actually lands.\n\nContribute to design and architecture across infrastructure and applications, with automation, performance, reliability, and security as first-class concerns.\n\nDrive infrastructure lifecycle at scale: provisioning, upgrades, and decommissioning across many isolated environments.\n\nEnsure infrastructure and applications meet or exceed enterprise compliance requirements, and design identity and access controls across platforms and services.\n\nPartner with enterprise customers on custom infrastructure requirements, translating their constraints into repeatable patterns rather than one-off work.\n\nRaise the bar through code and design review, and mentor SREs and product engineers on reliability practice.\n\nImprove and maintain infrastructure and process documentation.\n\nParticipate in and help evolve the on-call rotation, including how the team balances operational load against project work.\n\nQualifications and Experience\n10+ years in infrastructure, SRE, or platform engineering, including deep experience operating large-scale distributed systems in production.\n\nExpertise designing, analyzing, and troubleshooting distributed systems, with a track record of reliability decisions that held up under growth.\n\nDeep experience with at least one major public cloud provider, and the ability to reason across providers rather than within one.\n\nStrong command of container orchestration: cluster operation, workload scheduling, networking, and the failure modes that come with them.\n\nFluency with infrastructure as code, configuration management, and CI/CD pipeline design.\n\nStrong scripting and automation ability, and comfort reading and debugging application code in the languages your services are written in.\n\nExperience defining observability strategy - instrumentation, query languages, dashboards, and alert design that minimizes noise.\n\nDemonstrated ability to influence without authority and align multiple teams behind a technical direction.\n\nExperience operating under enterprise security and compliance frameworks.\n\nSolid understanding of Unix/Linux operating systems and networking fundamentals.","description_format":"text","description_chars":3360,"description_truncated":false,"requirements":{"experience_years_min":10,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[{"name":"United Kingdom","iso":"GB","kind":"country"}],"hiring_excludes":[],"relocation_offered":false,"industries":["Customer Data Platforms (CDP) & Reverse ETL"],"lifecycle":[{"event":"open","at":"2026-09-25T18:04:08Z"}],"liveness":{"score":17,"band":"cold","label":"Long shot","p_open":1,"p_active":0.48,"p_room":0.35,"age_days":14,"expected_fill_days":7,"reasons":["conf:17","win:tail"],"computed_at":"2026-10-01T05:45:00Z"},"pay":{"stated_usd_annual":250000,"is_top_pay":true},"html_url":"https://alion.io/job/metarouter-principal-site-reliability-engineer","json_url":"https://alion.io/job/metarouter-principal-site-reliability-engineer.json","meta":{"generated_at":"2026-10-01T10:41:35Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1841,"day_limit":5000,"remaining_today":3159,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}