{"id":1227979,"url":"https://alion.io/job/astika-software-technologies-site-reliability-engineer","title":"Site Reliability Engineer","company":{"id":3800837,"name":"Astika Software Technologies","domain":"astika.in","url":"https://alion.io/company/astika-software-technologies","size_band":null,"is_staffing_agency":true,"employer_type":"agency","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"DevOps","role_family":"DevOps","seniority":"senior","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Hyderabad, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":16000,"max_usd":37000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":42},"experience_years_min":8,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AWS","optional":false},{"name":"Azure","optional":false},{"name":"Bash","optional":false},{"name":"CI/CD","optional":false},{"name":"DNS","optional":false},{"name":"Docker","optional":false},{"name":"Grafana","optional":false},{"name":"Incident Management","optional":false},{"name":"Kubernetes","optional":false},{"name":"Linux","optional":false},{"name":"Prometheus","optional":false},{"name":"Python","optional":false},{"name":"SLI/SLO/SLA","optional":false},{"name":"Terraform","optional":false}],"status":"live","first_seen_at":"2026-09-22T09:57:47Z","employer_posted_date":null,"last_verified_at":"2026-09-22T09:57:47Z","board_verified":false,"closed_at":null,"days_open":9,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":9},"description":"Key Responsibilities : \n\n- Lead and drive Site Reliability Engineering (SRE) practices across production environments.\n\n- Design, implement, and maintain highly available, scalable, and resilient infrastructure and services.\n\n- Manage and optimize Kubernetes clusters and containerized workloads.\n\n- Build and maintain cloud infrastructure across AWS and/or Azure.\n\n- Develop and maintain Infrastructure as Code using Terraform.\n\n- Design, maintain, and improve CI/CD pipelines for reliable and efficient software delivery.\n\n- Implement automation to reduce manual operational effort and improve system reliability.\n\n- Establish and monitor SLIs, SLOs, and error budgets for critical services.\n\n- Develop and maintain monitoring, alerting, dashboards, and observability solutions using Prometheus and Grafana.\n\n- Lead production incident management, including incident response, coordination, and resolution.\n\n- Conduct detailed Root Cause Analysis (RCA) for production incidents and drive corrective and preventive actions.\n\n- Identify system bottlenecks, reliability risks, and opportunities for performance and availability improvements.\n\n- Implement proactive monitoring and capacity planning to prevent production issues.\n\n- Establish and improve operational processes, runbooks, and reliability standards.\n\n- Collaborate closely with Engineering, Development, QA, Security, and Product teams to improve application and infrastructure reliability.\n\n- Mentor SRE/DevOps engineers and provide technical leadership on reliability and automation initiatives.\n\n- Participate in on-call and production support activities as required.\n\nRequired Technical Skills : \n\n- 8+ years of experience in SRE, DevOps, Infrastructure Engineering, or a related role.\n\n- Strong hands-on experience with Kubernetes and Docker/containerized environments.\n\n- Strong experience with AWS and/or Azure cloud platforms.\n\n- Strong Linux administration and troubleshooting skills.\n\n- Hands-on experience with Terraform and Infrastructure as Code.\n\n- Strong understanding and experience with CI/CD practices and tools.\n\n- Experience with Prometheus, Grafana, and modern monitoring/observability practices.\n\n- Strong understanding of SLI, SLO, SLA, error budgets, and reliability engineering principles.\n\n- Experience with production incident management, troubleshooting, and RCA.\n\n- Strong scripting/automation skills using technologies such as Python, Bash, or similar.\n\n- Good understanding of networking, DNS, load balancing, security, and distributed systems.\n\n- Experience designing and operating highly available and fault-tolerant systems.\nSkills\nKubernetes, AWS, Azure, Terraform, Prometheus, Grafana, Python, Cloud Infrastructure, CI/CD Pipeline, Bash Scripting, Site Reliability","description_format":"text","description_chars":2762,"description_truncated":false,"requirements":{"experience_years_min":8,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-09-25T13:06:44Z"}],"liveness":{"score":44,"band":"fade","label":"Fading","p_open":1,"p_active":0.469,"p_room":0.945,"age_days":8,"expected_fill_days":24,"reasons":["seen:8","agency","velocity","win:mid"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/astika-software-technologies-site-reliability-engineer","json_url":"https://alion.io/job/astika-software-technologies-site-reliability-engineer.json","meta":{"generated_at":"2026-10-01T21:16:56Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":3911,"day_limit":5000,"remaining_today":1089,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}