{"id":1425986,"url":"https://alion.io/job/dynamo-senior-devops-engineer","title":"Senior DevOps Engineer","company":{"id":23844,"name":"Dynamo","domain":"dynamo.ai","url":"https://alion.io/company/dynamo","size_band":"51-200","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Rippling","truth_index":{"grade":"A","score":87,"open_postings":13,"ghost_share":0,"stale_share":0.538,"repost_share":0,"time_to_fill_p50_days":null,"computed_at":"2026-10-01T05:45:00Z"}},"role":"DevOps","role_family":"DevOps","seniority":"senior","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bengaluru, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":17500,"max_usd":40000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":42},"experience_years_min":5,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AI Agents","optional":false},{"name":"Amazon Aurora","optional":false},{"name":"Amazon EC2","optional":false},{"name":"Amazon EKS","optional":false},{"name":"Amazon S3","optional":false},{"name":"ArgoCD","optional":false},{"name":"AWS","optional":false},{"name":"CI/CD","optional":false},{"name":"External Secrets","optional":false},{"name":"GitHub Actions","optional":false},{"name":"GitOps","optional":false},{"name":"Grafana","optional":false},{"name":"HashiCorp Vault","optional":false},{"name":"Helm","optional":false},{"name":"IAM","optional":false},{"name":"Jenkins","optional":false},{"name":"Kubernetes","optional":false},{"name":"LLM Guardrails","optional":false},{"name":"OpenTelemetry","optional":false},{"name":"Platform Engineering","optional":false},{"name":"Prometheus","optional":false},{"name":"Python","optional":false},{"name":"Terraform","optional":false},{"name":"Thanos","optional":false},{"name":"Apache Kafka","optional":true},{"name":"ISO 27001","optional":true},{"name":"LLM","optional":true},{"name":"PostgreSQL","optional":true},{"name":"Redis","optional":true},{"name":"SOC 2","optional":true}],"status":"live","first_seen_at":"2026-09-28T19:28:43Z","employer_posted_date":"2026-09-28","last_verified_at":"2026-10-01T08:25:36Z","board_verified":true,"closed_at":null,"days_open":2,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":2},"description":"About Dynamo AI\nDynamo AI helps enterprises deploy AI systems that are reliable, secure, and production-ready. Our market-leading technical controls span AI evaluations, guardrails, agentic risk management, and observability. Backed by world-class talent, we partner with innovative, highly-regulated global organizations - across financial services, government, and beyond - to deploy meaningful AI use cases at scale, securely and compliantly.\nAbout the Role\nWe are looking for a Senior DevOps Engineer to help build, scale, and operate the infrastructure powering our AI platform. This is a hands-on, high-ownership role. We are looking for someone who can work independently, solve complex infrastructure problems, and thrive in a fast-paced startup environment. You should be comfortable designing systems, automating processes, troubleshooting production issues, and continuously improving reliability, scalability, and cost efficiency.\nWhat You'll Own\nDesign, build, and operate highly available production infrastructure on AWS, with strong expertise in EKS, EC2, VPC, S3, RDS/Aurora, IAM, ECR, ElastiCache, Load Balancers, and other core AWS services.\nBuild and improve CI/CD and release automation using Jenkins, GitHub Actions, Helm, ArgoCD, and GitOps.\nManage infrastructure using Terraform and Infrastructure as Code principles.\nBuild and operate Kubernetes platforms at production scale, including cluster management, upgrades, autoscaling, networking, security, and troubleshooting.\nRun and operate AI/ML workloads in production, with a strong understanding of the infrastructure challenges associated with AI systems.\nDeploy, scale, monitor, and optimize AI inference and model-serving workloads across Kubernetes and cloud infrastructure.\nWork with GPU-based workloads, including GPU scheduling, utilization, autoscaling, capacity planning, and optimization.\nDrive infrastructure efficiency by balancing performance, reliability, scalability, and cost across AI workloads.\nOwn monitoring, logging, and observability using tools such as Prometheus, Grafana, Thanos, and OpenTelemetry.\nImplement secure secrets management using technologies such as HashiCorp Vault and External Secrets Operator.\nDevelop automation and internal tooling using Python and Bash.\nDrive improvements around reliability, security, scalability, performance, and infrastructure cost.\nParticipate in production incidents, root-cause analysis, and drive long-term fixes rather than short-term workarounds.\nWork closely with Engineering, ML/AI, Security, and Product teams to solve infrastructure and platform challenges.\nWhat We're Looking For\n5+ years of strong hands-on experience in DevOps, SRE, Platform Engineering, or Cloud Infrastructure.\nStrong production experience with AWS and Kubernetes/EKS.\nProven experience running AI/ML workloads or GPU-based workloads in production is highly valuable.\nStrong understanding of CI/CD, Infrastructure as Code, GitOps, observability, and cloud security.\nExcellent scripting and automation skills in Python and Bash.\nExperience operating production systems and troubleshooting complex infrastructure issues independently.\nStrong understanding of scaling, performance optimization, resource utilization, and cost management, particularly for compute-intensive workloads.\nStrong ownership mindset with the ability to take a problem from design to production.\nExperience working in a startup or fast-moving engineering environment is highly valued.\nStrong communication skills and the ability to work effectively across teams.\nNice to Have\nExperience with AI inference platforms, model serving, LLM infrastructure, or ML platforms.\nExperience with GPU infrastructure such as NVIDIA GPUs and Kubernetes GPU scheduling.\nExperience with multi-region or highly distributed systems.\nExperience with SOC 2, ISO 27001, or other security/compliance requirements.\nExperience with PostgreSQL, MongoDB, Redis, Kafka, or similar distributed systems.\nThe Kind of Engineer We Want\nWe're looking for someone who builds, automates, and takes ownership - not someone who simply operates existing infrastructure.\nYou should be comfortable with ambiguity, willing to dive deep into production problems, and constantly looking for ways to make our platform more reliable, secure, scalable, and cost-efficient.\nMost importantly, you should understand that AI infrastructure has a different set of operational challenges. We want someone who can help us run AI systems efficiently at scale - making the right trade-offs between GPU utilization, performance, reliability, scalability, and cost.\nThis is a high-ownership role for someone who wants to make a meaningful impact on the infrastructure behind an AI platform as we scale.","description_format":"text","description_chars":4739,"description_truncated":false,"requirements":{"experience_years_min":5,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Artificial Intelligence","LLM & Generative AI","AI Evaluation & Observability","AI Governance & Compliance"],"lifecycle":[{"event":"open","at":"2026-09-28T23:41:45Z"}],"liveness":{"score":63,"band":"ok","label":"Likely open","p_open":1,"p_active":0.632,"p_room":1,"age_days":2,"expected_fill_days":42,"reasons":["conf:9","stale_co","velocity","win:early"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/dynamo-senior-devops-engineer","json_url":"https://alion.io/job/dynamo-senior-devops-engineer.json","meta":{"generated_at":"2026-10-01T19:12:17Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1659,"day_limit":5000,"remaining_today":3341,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}