{"id":1511159,"url":"https://alion.io/job/algoleap-mlopsllm-infrastructure-engineer","title":"MLOps/LLM Infrastructure Engineer","company":{"id":3800168,"name":"Algoleap","domain":"algoleap.com","url":"https://alion.io/company/algoleap-technologies","size_band":"1001-5000","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"AI/ML","role_family":"AI/ML","seniority":"senior","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Mumbai, India","India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":21000,"max_usd":43000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":29},"experience_years_min":7,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"CI/CD","optional":false},{"name":"Docker","optional":false},{"name":"Gemma","optional":false},{"name":"Kubernetes","optional":false},{"name":"LLM","optional":false},{"name":"Mistral","optional":false},{"name":"Triton","optional":false},{"name":"vLLM","optional":false},{"name":"CUDA","optional":true},{"name":"CUDA Toolkit","optional":true},{"name":"Grafana","optional":true},{"name":"Helm","optional":true},{"name":"MLFlow","optional":true},{"name":"Prometheus","optional":true},{"name":"Quantization","optional":true},{"name":"TensorRT","optional":true},{"name":"TensorRT-LLM","optional":true}],"status":"live","first_seen_at":"2026-09-30T07:00:32Z","employer_posted_date":null,"last_verified_at":"2026-09-30T07:00:32Z","board_verified":false,"closed_at":null,"days_open":2,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":2},"description":"Role Overview:\n\nResponsible for hosting, deploying, and operating open-weight LLMs within a sovereign cloud environment. The role focuses on GPU infrastructure, model serving, performance optimization, and reliable model lifecycle management.\n\nKey Responsibilities:\n\n- Deploy and operate models such as GPT, LLaMA, Gemma, Mistral, and other product/open-weight models within sovereign cloud.\n\n- Manage GPU provisioning, capacity planning, utilization, and performance optimization.\n\n- Implement LLM inference serving using platforms such as vLLM, Triton, or similar frameworks.\n\n- Build model deployment, versioning, rollback, and lifecycle management processes.\n\n- Develop MLOps pipelines for model packaging, testing, deployment, and monitoring.\n\n- Monitor latency, throughput, GPU utilization, availability, and inference costs.\n\n- Implement scalable and highly available model-serving infrastructure using Kubernetes and containers.\n\n- Work closely with platform, security, and gateway teams to ensure secure model access and governance.\n\n- Troubleshoot production issues across GPU, inference, Kubernetes, networking, and model-serving layers.\n\nTech Stack:\n\n- MLOps, LLM infrastructure, model serving, GPU-based inference, Kubernetes, vLLM, NVIDIA Triton, TensorRT-LLM, LLaMA, Gemma, Mistral, GPT, Docker, CI/CD, model registries, observability, quantization, batching, caching, GPU memory management, inference optimization, private/sovereign-cloud environments.\n\nPreferred Skills:\n\n- NVIDIA GPUs, CUDA, Helm, Prometheus/Grafana, MLflow, automated model deployment pipelines.\n\nSkills\nMLOps, Kubernetes, Docker, CUDA, CI/CD, LLM, GPU, LLama, Prometheus, IT Infrastructure","description_format":"text","description_chars":1676,"description_truncated":false,"requirements":{"experience_years_min":7,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Science & Engineering","Engineering Services"],"lifecycle":[{"event":"open","at":"2026-09-30T08:00:00Z"}],"liveness":{"score":90,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.903,"p_room":1,"age_days":1,"expected_fill_days":30,"reasons":["seen:1","velocity","win:early"],"computed_at":"2026-10-02T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/algoleap-mlopsllm-infrastructure-engineer","json_url":"https://alion.io/job/algoleap-mlopsllm-infrastructure-engineer.json","meta":{"generated_at":"2026-10-03T02:49:36Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":2788,"day_limit":5000,"remaining_today":2212,"minute_limit":60,"resets_at":"2026-10-04T00:00:00Z"}}}