{"id":1512257,"url":"https://alion.io/job/backblaze-site-reliability-engineer-iii-dba","title":"Site Reliability Engineer III (DBA)","company":{"id":37551,"name":"Backblaze","domain":"backblaze.com","url":"https://alion.io/company/backblaze","size_band":"201-500","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Greenhouse","truth_index":{"grade":"A","score":95,"open_postings":15,"ghost_share":0,"stale_share":0,"repost_share":0,"time_to_fill_p50_days":115,"computed_at":"2026-10-01T05:45:00Z"}},"role":"DevOps","role_family":"DevOps","seniority":"middle","employment_type":null,"work_mode":"remote","remote_scope":"stated_countries","remote_scope_basis":"posting_text","remote_working_hours":null,"hiring_geo_confidence":"structured","locations":[],"countries":[],"hiring_countries":["US"],"hiring_countries_total":1,"salary":{"min":125000,"max":150000,"currency":"USD","period":"year","gross":null,"usd_annual":150000},"salary_estimate":null,"experience_years_min":6,"visa_sponsorship":false,"relocation_package":false,"has_equity":true,"technologies":[{"name":"Ansible","optional":false},{"name":"Bash","optional":false},{"name":"Cassandra","optional":false},{"name":"CI/CD","optional":false},{"name":"Configuration Management","optional":false},{"name":"Docker","optional":false},{"name":"Error Budget","optional":false},{"name":"Grafana","optional":false},{"name":"ITIL","optional":false},{"name":"Jenkins","optional":false},{"name":"kubectl","optional":false},{"name":"Kubernetes","optional":false},{"name":"Linux","optional":false},{"name":"MySQL","optional":false},{"name":"Nomad","optional":false},{"name":"Prometheus","optional":false},{"name":"Python","optional":false},{"name":"SLI/SLO/SLA","optional":false},{"name":"SQL","optional":false},{"name":"SRE","optional":false},{"name":"Terraform","optional":false},{"name":"Vault","optional":false},{"name":"AWS","optional":true},{"name":"Azure","optional":true},{"name":"GCP","optional":true}],"status":"live","first_seen_at":"2026-09-30T01:54:58Z","employer_posted_date":"2026-09-30","last_verified_at":"2026-10-01T18:03:08Z","board_verified":true,"closed_at":null,"days_open":1,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":1},"description":"About Backblaze\nBackblaze is the object storage leader in the open cloud movement, fueling customer success with cloud storage built purposefully to unlock budgets, unburden administrators, and unleash innovators. Together with our partners, we’re helping customers break free from the restrictive, overpriced legacy solutions that hold them back, and blaze forward with the full power of the open cloud in their hands.\nFounded in 2007, we scaled the business with less than $3 million in outside funding until 2021, when we did a traditional IPO on the Nasdaq stock exchange. Today, Backblaze generates over $100m in revenue and is the leading specialized storage cloud - managing over three billion gigabytes of data storage for 500K+ customers in 175+ countries, including businesses, developers, IT professionals, and individuals.\nBut while there is a lot to celebrate in our past, there is almost as much opportunity ahead of us. We’re seeking a Site Reliability Engineer III (DBA) to join our team!\nAbout the Role:\nIndividuals fulfilling this role will be responsible for ensuring the stability, scalability, and reliability of our production database systems, primarily Vitess (distributed MySQL) and Cassandra, alongside the rest of our production services and infrastructure. This role carries the same on-call, incident response, and service ownership expectations as other SRE IIIs, with database systems serving as the area of deepest technical ownership.\nBecause our SRE Database Engineering function is new, this role will also help establish its operational foundation by designing database architecture and developing the runbooks, escalation guidance, procedures, and training materials that our Level 1 and Level 2 SRE Database Engineers will use as they onboard. The ideal candidate will have strong experience with production database systems, Linux, automation, distributed systems, Kubernetes, observability, and incident response, with a proactive approach to reliability and operational excellence.\nWhat You'll Do:\n Database Architecture & Administration:Design, deploy, and own highly available database architecture for Vitess (distributed MySQL) and Cassandra\nEstablish and document operational procedures, runbooks, and escalation guidance for Level 1 and Level 2 SRE Database Engineers\nOptimize database performance through query tuning, indexing strategies, schema design, and capacity planning\nOwn database backup, recovery, replication, and disaster recovery strategies\nPerform and validate disaster recovery testing and database recovery procedures\nDrive database security, access control, patching, hardening, and compliance practices\nPartner with the DBA and Data Infrastructure teams on resharding, capacity planning, replication, and architecture decisions for sharded MySQL environments\n\nService Reliability & Operations:Support the availability and durability of critical services across production environments\nMonitor service health using SLIs, SLOs, error budgets, monitoring, logging, and alerting platforms\nPartner with service owners to define and improve SLIs, SLOs, error budget policies, and alerting\nParticipate in on-call rotations, incident response, root cause analysis, and post-incident reviews\nServe as an escalation point for complex database production incidents\nFollow established ITIL/OSS processes including incident, change, problem, and capacity management\nTake ownership of operational issues and drive projects from problem discovery through resolution\n\nAutomation & Tooling:Develop automation for common operational and database administration tasks to reduce manual intervention and operational toil\nContribute to monitoring, logging, and alerting frameworks including Prometheus, Grafana, Catchpoint, and ELK\nHelp integrate operational runbooks and incident response workflows with FireHydrant\nWork with CI/CD pipelines, configuration management, and infrastructure as code tools including Terraform, Ansible, and Jenkins\nDevelop scripts using Bash, Python, Go, or similar technologies to improve reliability and operational efficiency\nOperate and troubleshoot containerized production environments using Kubernetes and Docker\nWork within Kubernetes and Vitess environments using technologies such as kubectl, mysqlsh, and Vitess keyspaces\n\n Project Management:Lead Production Readiness Reviews (PRRs) for functionality being handed off from engineering partner teams\nSupport the operational readiness of new database-backed services before they enter production\nBuild training plans, onboarding materials, and technical documentation for new Level 1 and Level 2 SRE Database Engineers\nPartner with Engineering, Product, Operations, and DBA/Data Infrastructure teams on reliability initiatives\nAssist with capacity planning, disaster recovery exercises, database migrations, and infrastructure projects\nWork with vendors and service providers to troubleshoot service issues and track SLA performance\nIdentify opportunities for automation and process efficiency\n\nIncident responseRespond to and resolve production database, infrastructure, and service incidents\nTroubleshoot and escalate database, Linux, networking, application, and infrastructure issues as needed\nParticipate in the on-call rotation and serve as an escalation point for database-related incidents\nLead or contribute to root cause analysis and post-incident reviews\nIdentify recurring issues and develop long-term corrective actions to improve reliability\n\nWhat we value:A proactive mindset with a can-do attitude\nSomeone who can work independently, take ownership, and drive complex technical problems through resolution\nSomeone who steps up, supports teammates, mentors others, and shares knowledge freely\nStrong problem-solving skills and a willingness to learn new technologies\nCuriosity, reliability, and a desire to improve the reliability and scalability of production systems\n\nRequired Qualifications:\n6-8 years of experience in site reliability engineering, systems engineering, infrastructure operations, database engineering, or similar roles, with meaningful experience supporting production database systems.\nDeep hands-on experience with MySQL and distributed or sharded database systems.\nExperience with Vitess in a production environment strongly preferred.\nExperience administering and supporting NoSQL databases such as Cassandra.\nExperience designing high-availability database architecture, replication topology, backup strategies, and disaster recovery processes.\nStrong SQL skills, including query performance analysis, indexing, schema design, and troubleshooting.\nSolid Linux systems administration and troubleshooting skills.\nExperience with security-focused operations including patching, system hardening, access controls, and vulnerability remediation.\nStrong understanding of service reliability concepts including monitoring, alerting, incident response, root cause analysis, SLIs, SLOs, and error budgets.\nExperience working with containers and orchestration platforms including Kubernetes and Docker.\nComfortable operating in Kubernetes and Vitess environments using tools such as kubectl, mysqlsh, and Vitess keyspaces.\nExperience with infrastructure and configuration management technologies including Terraform, Ansible, Jenkins, and HashiCorp products such as Vault and Nomad.\nProficiency in at least one scripting language such as Python, Bash, or Go.\nExperience establishing operational procedures, runbooks, documentation, and escalation processes.\nExperience mentoring, training, or helping onboard engineers into complex technical environments.\nExperience in SaaS, cloud services, service provider, or large-scale distributed systems environments preferred.\nExperience with AWS, GCP, Azure, or similar cloud platforms preferred.\nFamiliarity with ITIL/OSS practices and SLA/SLO management preferred.\nBachelor’s degree in Computer Science, Engineering, or a related field, or equivalent professional experience.\nBackblaze Perks:\nHealthcare for family, including dental and vision\nCompetitive compensation and 401K\nRSU grants for full-time employees\nESPP program\nFlexible vacation policy\nMaternity & paternity leave\nMacBook Pro to use for work, plus a generous stipend to personalize your workstation\nChildcare bonus (human children only)\nFertility treatment and support\nLearning & development program\nCommuter benefits\nCulture that supports a healthy work-life balance\nTo provide greater transparency to candidates, we share base pay ranges for all US-based job postings regardless of state. We set standard base pay ranges for all roles based on function, level, and country location, benchmarked against similar-stage growth companies. Final offer amounts are determined by multiple factors, including candidate location, skills, depth of work experience, and relevant licenses/credentials, and may vary from the amounts listed below.\nThe expected salary range for this role is - $125,000 - $150,000.\nAt Backblaze, we value being fair and good to our customers, partners, and employees. That’s why diversity, equity, and inclusion are at the core of our values. We are committed to fostering a workforce where all employees feel a sense of belonging regardless of race, ethnicity, nationality, gender, sexual orientation, age, religion, socio-economic status, ability, veteran status, and education. We believe that our dedication to cultivating a diverse workspace not only allows us to better serve our customers in over 175 countries but further reinforces our commitment to doing the right thing. We are proud to be an Equal Opportunity Employer.\nTo understand more about the data we collect and process as part of your application, please view our Backblaze Employee Privacy Notice.","description_format":"text","description_chars":9762,"description_truncated":false,"requirements":{"experience_years_min":6,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":false},"security_clearance":false,"languages":[]},"benefits":["401k plan","Apple Macbook","Equity","Flexible vacation policy"],"hiring_locations":[{"name":"United States","iso":"US","kind":"country"}],"hiring_excludes":[],"relocation_offered":false,"industries":["Cloud Storage & File Sharing"],"lifecycle":[{"event":"open","at":"2026-09-30T08:22:31Z"}],"liveness":{"score":90,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.903,"p_room":1,"age_days":1,"expected_fill_days":115,"reasons":["conf:11","velocity","win:early"],"computed_at":"2026-10-01T05:45:00Z"},"pay":{"stated_usd_annual":150000,"is_top_pay":true},"html_url":"https://alion.io/job/backblaze-site-reliability-engineer-iii-dba","json_url":"https://alion.io/job/backblaze-site-reliability-engineer-iii-dba.json","meta":{"generated_at":"2026-10-01T20:00:01Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":2702,"day_limit":5000,"remaining_today":2298,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}