{"id":1172285,"url":"https://alion.io/job/base14-site-reliability-engineer","title":"Site Reliability Engineer","company":{"id":2673873,"name":"Base14","domain":"base14.io","url":"https://alion.io/company/base14","size_band":null,"is_staffing_agency":false,"is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"DevOps","role_family":"DevOps","seniority":"junior","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bengaluru, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":null,"experience_years_min":2,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AWS","optional":false},{"name":"Azure","optional":false},{"name":"CI/CD","optional":false},{"name":"Datadog","optional":false},{"name":"GCP","optional":false},{"name":"GitOps","optional":false},{"name":"Go","optional":false},{"name":"Grafana","optional":false},{"name":"Kubernetes","optional":false},{"name":"Linux","optional":false},{"name":"OpenTelemetry","optional":false},{"name":"Platform Engineering","optional":false},{"name":"Prometheus","optional":false},{"name":"Python","optional":false},{"name":"SRE","optional":false},{"name":"Terraform","optional":false}],"status":"live","first_seen_at":"2026-09-24T08:30:31Z","employer_posted_date":null,"last_verified_at":"2026-09-24T08:30:31Z","board_verified":false,"closed_at":null,"days_open":0,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":0},"description":"Responsibilities:\nDesign, build, and operate highly reliable cloud infrastructure supporting modern distributed applications. Develop automation and platform tooling using Python and Infrastructure as Code to improve engineering productivity and operational efficiency.\nBuild and maintain scalable cloud infrastructure across AWS, GCP, or Azure using Terraform and modern cloud-native practices.\nDesign and enhance observability solutions using Prometheus, Grafana, distributed tracing, and related monitoring technologies. Improve deployment workflows through CI/CD automation, GitOps practices, and infrastructure automation.\nParticipate in incident response, troubleshoot production issues, and continuously improve platform reliability and resiliency.\nBuild intelligent automation around monitoring, alerting, diagnostics, and operational workflows. Contribute to the design and development of next-generation observability and reliability products.\nCollaborate closely with the founding team on architecture decisions, platform design, and product development.\nContinuously improve scalability, reliability, security, and developer experience while working in a highly collaborative, ownership-driven engineering culture.\nRequirements:\nBachelor's degree in computer science, engineering, or a related field. 2-4 years of experience in platform engineering, cloud engineering, site reliability engineering, or infrastructure engineering.\nStrong hands-on experience with at least one cloud platform (AWS, GCP, or Azure) in production environments. Strong proficiency in Infrastructure as Code using Terraform.\nGood programming skills in Python or any modern programming language, with a focus on automation and tooling.\nHands-on experience with Kubernetes and containerized workloads.\nExperience building automation for infrastructure provisioning, deployments, operational workflows, or platform tooling.\nStrong understanding of Linux systems, networking fundamentals, and cloud-native architectures. Experience working with observability and monitoring tools such as Prometheus, Grafana, Datadog, or similar.\nGood understanding of CI/CD pipelines, GitOps workflows, and infrastructure automation. Experience working on large-scale distributed systems and production infrastructure is preferred.\nExposure to OpenTelemetry, security best practices, and open-source technologies is an added advantage. Excellent problem-solving, debugging, and communication skills, with the ability to work in a fast-paced startup environment.","description_format":"text","description_chars":2529,"description_truncated":false,"requirements":{"experience_years_min":2,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":false},"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-09-24T08:30:31Z"}],"liveness":{"score":86,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.86,"p_room":1,"age_days":0,"expected_fill_days":14,"reasons":["seen:0","win:early","comp:junior"],"computed_at":"2026-09-25T01:16:06Z"},"pay":null,"html_url":"https://alion.io/job/base14-site-reliability-engineer","json_url":"https://alion.io/job/base14-site-reliability-engineer.json","meta":{"generated_at":"2026-09-25T01:16:06Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1239,"day_limit":5000,"remaining_today":3761,"minute_limit":60,"resets_at":"2026-09-26T00:00:00Z"}}}