{"id":1243103,"url":"https://alion.io/job/opensource-technologies-cloud-engineer-kubernetes","title":"Cloud Engineer Kubernetes","company":{"id":116582,"name":"OpenSource Technologies","domain":"ost.agency","url":"https://alion.io/company/ost-2","size_band":null,"is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"DevOps","role_family":"DevOps","seniority":"senior","employment_type":"contractor","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Singapore"],"countries":["SG"],"hiring_countries":[],"hiring_countries_total":0,"salary":{"min":8000,"max":11000,"currency":"SGD","period":"month","gross":true,"usd_annual":103356},"salary_estimate":null,"experience_years_min":5,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"ArgoCD","optional":false},{"name":"Bash","optional":false},{"name":"C++","optional":false},{"name":"Datadog","optional":false},{"name":"DNS","optional":false},{"name":"GitOps","optional":false},{"name":"Grafana","optional":false},{"name":"Helm","optional":false},{"name":"Incident Management","optional":false},{"name":"Kubernetes","optional":false},{"name":"Linux","optional":false},{"name":"OpenSearch","optional":false},{"name":"Prometheus","optional":false},{"name":"Python","optional":false},{"name":"Service Mesh","optional":false},{"name":"Splunk","optional":false},{"name":"TCP/IP","optional":false},{"name":"Ubuntu","optional":false}],"status":"live","first_seen_at":"2026-09-22T00:00:00Z","employer_posted_date":null,"last_verified_at":"2026-09-22T00:00:00Z","board_verified":false,"closed_at":null,"days_open":13,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":13},"description":"Role Overview\nWe are looking for experienced Kubernetes & Site Reliability Engineers to support highly scalable, business-critical production platforms for a global technology customer in Singapore.\nThe role requires strong hands-on expertise in Kubernetes, Linux, production reliability, automation, observability, incident management and troubleshooting of distributed systems\nCandidates should be comfortable operating large-scale production environments where availability, performance, automation and operational excellence are critical.\nKey Responsibilities\nOperate, maintain and troubleshoot large-scale Kubernetes-based production environments\nEnsure reliability, scalability, availability and performance of critical services. \nInvestigate complex production issues and perform detailed root-cause analysis. \nParticipate in incident response and drive permanent corrective actions. \nAutomate repetitive operational activities and improve platform reliability. \nBuild and improve monitoring, alerting, logging and observability frameworks. \nDefine and track SLIs, SLOs and operational reliability metrics\nSupport Kubernetes upgrades, configuration changes, patching and platform improvements. \nWork closely with application engineering, infrastructure, platform, security and DevOps teams. \nPerform capacity planning, performance tuning and reliability improvements. \nDevelop and maintain operational runbooks, automation scripts and troubleshooting documentation. \nParticipate in production readiness reviews and ensure applications meet operational standards. \nMandatory Skills\nStrong hands-on experience with \nKubernetes administration and troubleshooting\nStrong understanding of Kubernetes architecture, including: \nPods \nDeployments \nStatefulSets \nServices \nIngress \nConfigMaps / Secrets \nRBAC \nStorage \nNetworking \nStrong \nLinux systems administration and troubleshooting skills. \nGood understanding of networking concepts such as DNS, TCP/IP, load balancing and service connectivity. \nStrong understanding of \nSite Reliability Engineering principles\nExperience supporting large-scale, high-availability production systems. \nStrong incident management and RCA experience. \nHands-on scripting/automation experience using \nPython, Bash/Shell or similar\nExperience with monitoring and observability tools such as \nPrometheus, Grafana, Splunk, ELK/OpenSearch, Datadog or equivalent \nHelm or similar Kubernetes package/deployment management tools. \nGitOps experience using tools such as Argo CD or Flux. \nKnowledge of service mesh concepts. \nExperience with container security and Kubernetes security practices. \nExperience with cloud or private-cloud infrastructure. \nFamiliarity with distributed systems and microservices architectures. \nExposure to performance engineering and capacity management. \nExperience working in globally distributed engineering environments. \nSkills\nOperational Efficiency, RDS, systems reliability, Oracle Alerts, Scalability, Automated Operation Monitoring, Ubuntu, Patch Management, Reliability Requirements, Root Cause Analysis, Availability Management, Technical Consultation, Routing Protocols, Incident Handling, Configuration Changes, C++","description_format":"text","description_chars":3184,"description_truncated":false,"requirements":{"experience_years_min":5,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Mobile App Development Services","Web Development Services","Custom Software Development","Digital Marketing"],"lifecycle":[{"event":"open","at":"2026-09-25T16:15:06Z"}],"visa":[],"liveness":{"score":72,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.802,"p_room":0.9,"age_days":12,"expected_fill_days":24,"reasons":["seen:12","velocity","win:mid"],"computed_at":"2026-10-04T05:45:00Z"},"pay":{"stated_usd_annual":103356,"is_top_pay":false},"html_url":"https://alion.io/job/opensource-technologies-cloud-engineer-kubernetes","json_url":"https://alion.io/job/opensource-technologies-cloud-engineer-kubernetes.json","meta":{"generated_at":"2026-10-05T01:09:56Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1473,"day_limit":5000,"remaining_today":3527,"minute_limit":60,"resets_at":"2026-10-06T00:00:00Z"}}}