{"id":1607599,"url":"https://alion.io/job/penguin-solutions-sr-software-engineer-ai-platform","title":"Sr. Software Engineer - AI Platform","company":{"id":178085,"name":"Penguin Solutions","domain":"penguinsolutions.com","url":"https://alion.io/company/penguin-solutions","size_band":"201-500","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"SuccessFactors","truth_index":null},"role":"Backend","role_family":"Backend","seniority":"senior","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bengaluru, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":17500,"max_usd":45000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":123},"experience_years_min":7,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AI Agents","optional":false},{"name":"CI/CD","optional":false},{"name":"Docker","optional":false},{"name":"HPC","optional":false},{"name":"Kubernetes","optional":false},{"name":"LLM","optional":false},{"name":"Platform Engineering","optional":false},{"name":"Python","optional":false},{"name":"RAG","optional":false},{"name":"Rest API","optional":false},{"name":"Triton Inference Server","optional":false},{"name":"vLLM","optional":false},{"name":"AWS","optional":true},{"name":"AWS Bedrock","optional":true},{"name":"Azure","optional":true},{"name":"LangSmith","optional":true},{"name":"MLFlow","optional":true},{"name":"OpenTelemetry","optional":true},{"name":"Vertex AI","optional":true}],"status":"live","first_seen_at":"2026-09-11T07:00:00Z","employer_posted_date":"2026-09-11","last_verified_at":"2026-10-02T08:59:05Z","board_verified":true,"closed_at":null,"days_open":21,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":21},"description":"At Penguin Solutions (Nasdaq: PENG) - The AI Factory Platform Company - we’re building a team of innovators who thrive on collaboration, creativity, and the opportunity to help shape the future of AI. As part of the AI technology revolution, our teams design, build, deploy, and manage AI factories for enterprises, sovereign AI initiatives, and neocloud providers worldwide.\nHeadquartered in Silicon Valley, California, Penguin Solutions operates globally through a network of R&D, manufacturing, and sales locations. For nearly three decades, we have operated at the intersection of memory and AI/HPC infrastructure. That engineering expertise positions us to power the next generation of AI workloads, from training to inference and agentic AI at scale.\nPenguin Solutions brings together differentiated infrastructure software, advanced memory, compute systems, end-to-end services, and industry-leading partner solutions in a full-stack AI factory platform designed to help customers deploy and scale AI workloads with speed and precision.\nAt Penguin Solutions, we value ideas over hierarchy and believe in servant leadership, where leaders enable teams to do their best work. We empower employees to take ownership, drive innovation, and grow through challenging work, continuous learning, and exposure to advanced AI tools and technologies. With flexibility where it matters and a strong focus on outcomes, Penguin Solutions is a place to do your best work, grow your career, and make a meaningful impact.\nJob Overview\nWe are seeking a Senior AI Platform Engineer to build and operate the production AI platform that powers ClusterWareAI. In this role, you will develop the infrastructure, services, and tooling that enable our AI Operational Agent to reliably deploy, retrieve knowledge, execute workflows, and operate at scale in enterprise environments.\nYou will work closely with AI engineers, platform engineers, and software developers to build secure, scalable, observable, and highly available AI services for production deployments.\nResponsibilities\nBuild and operate the production AI platform that supports model serving, inference, and AI services. \nDevelop scalable retrieval, knowledge management, and data ingestion pipelines to support AI-driven operations. \nBuild APIs, platform services, and automation that integrate AI capabilities into ClusterWareAI. \nImplement AI observability, monitoring, evaluation, and operational tooling for production AI services. \nOptimize inference performance, scalability, reliability, and operational efficiency. \nBuild and maintain CI/CD pipelines and deployment automation for AI applications. \nPartner with AI, platform, and software engineering teams to deploy AI capabilities into production. \nContribute to platform security, reliability, and operational excellence. \nQualifications\nBachelor's degree in Computer Science, Engineering, or a related field. \n7+ years of software engineering, platform engineering, or infrastructure engineering experience. \nStrong programming skills in Python. \nExperience developing and operating production AI or LLM-powered applications. \nExperience with Kubernetes, Docker, and cloud-native architectures. \nExperience with model serving technologies such as vLLM, NVIDIA Triton Inference Server, or similar platforms. \nExperience building REST APIs, microservices, and distributed systems. \nFamiliarity with vector databases, retrieval systems, and RAG architectures. \nPreferred Qualifications\nExperience implementing monitoring, observability, and operational tooling for AI services. \nExperience with managed AI platforms such as Azure AI Foundry, AWS Bedrock, or Google Vertex AI. \nExperience with AI engineering tools such as LangSmith, MLflow, OpenTelemetry, or similar technologies. \nExperience deploying GPU-based AI workloads in Kubernetes environments. \nExperience integrating AI services with enterprise applications, APIs, or infrastructure platforms. \nBackground in enterprise infrastructure, cloud platforms, or systems management software.\nLocation\nHybrid - Bangalore\nTravel\nAs needed","description_format":"text","description_chars":4096,"description_truncated":false,"requirements":{"experience_years_min":7,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":false},"security_clearance":false,"languages":[]},"benefits":["Continuous learning"],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Artificial Intelligence","Education","LLM & Generative AI","High-Performance Computing"],"lifecycle":[{"event":"open","at":"2026-10-01T19:57:58Z"}],"liveness":{"score":76,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.849,"p_room":0.9,"age_days":20,"expected_fill_days":47,"reasons":["conf:9","win:mid"],"computed_at":"2026-10-02T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/penguin-solutions-sr-software-engineer-ai-platform","json_url":"https://alion.io/job/penguin-solutions-sr-software-engineer-ai-platform.json","meta":{"generated_at":"2026-10-03T03:59:24Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4000,"day_limit":5000,"remaining_today":1000,"minute_limit":60,"resets_at":"2026-10-04T00:00:00Z"}}}