{"id":1449256,"url":"https://alion.io/job/trend-micro-staffsr-ml-infrastructure-platform-engineer","title":"Staff/Sr. ML Infrastructure / Platform Engineer","company":{"id":20752,"name":"Trend Micro","domain":"trendmicro.com","url":"https://alion.io/company/trend-micro","size_band":"1001-5000","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Workday","truth_index":{"grade":"B","score":80,"open_postings":42,"ghost_share":0,"stale_share":0.786,"repost_share":0,"time_to_fill_p50_days":50,"computed_at":"2026-10-01T05:45:00Z"}},"role":"DevOps","role_family":"DevOps","seniority":"staff","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Taipei, Taiwan"],"countries":["TW"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":49000,"max_usd":114000,"period":"year","method":"global_role_cell_scaled_by_country","sample_n":603},"experience_years_min":null,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AWS","optional":false},{"name":"GCP","optional":false},{"name":"Grafana","optional":false},{"name":"Helm","optional":false},{"name":"Kubernetes","optional":false},{"name":"KV Cache","optional":false},{"name":"LLM","optional":false},{"name":"Prometheus","optional":false},{"name":"SLI/SLO/SLA","optional":false},{"name":"Terraform","optional":false},{"name":"Terragrunt","optional":false},{"name":"CI/CD","optional":true},{"name":"Fine-tuning","optional":true},{"name":"LoRA","optional":true},{"name":"MLFlow","optional":true},{"name":"NVIDIA NIM","optional":true},{"name":"PEFT","optional":true},{"name":"SGLang","optional":true},{"name":"Speculative Decoding","optional":true},{"name":"Transformers","optional":true}],"status":"live","first_seen_at":"2026-09-29T07:28:08Z","employer_posted_date":"2026-09-29","last_verified_at":"2026-10-01T06:28:07Z","board_verified":true,"closed_at":null,"days_open":2,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":2},"description":"Join Trend ‧ Join New Generation\n趨勢科技 - 全球雲端資安領航者 / 全亞洲最大軟體公司 / 企業版圖橫跨五大洲 / 趨勢全球研發基地在台灣\n===============================================================\nAbout the Role\nWe are building a production-grade, GPU-accelerated LLM serving platform that powers multiple AI products at enterprise scale. You will be responsible for designing, building, and operating the infrastructure that serves large language models - from raw Kubernetes cluster management to multi-GPU inference optimization and autoscaling.\nRequired Qualifications\nModel Serving & Inference\nOperate multi-model LLM serving infrastructure \nTune autoscaling policies to balance GPU cost and latency SLAs \nKubernetes & GPU Infrastructure\nOperate production K8s clusters with NVIDIA GPU nodes \nHandle GPU node lifecycle: NVIDIA driver setup \nInfrastructure as Code\nWrite and maintain Terraform/Terragrunt modules for AWS/GCP cloud \nPackage platform components and model deployments as Helm charts \nManage multi-environment configurations \nObservability & Performance\nMaintain monitoring stack: Prometheus, Grafana, \nBuild dashboards for GPU utilization, KV cache occupancy, TTFT/ITL latency, and cost per token \nSet up alerting for SLA violations and OOM events \nBonus Skills\nThese are not required, but candidates with these skills will stand out.\nLoRA / PEFT fine-tuning workflows \nMLflow for experiment tracking, model registry, and automated adapter deployment \nExperience building LoRA adapter CI/CD pipelines (training → registry → serving) \nExperience with alternative inference frameworks such as SGLang or NVIDIA NIM, including deep Parameter Tuning for Continuous Batching, KV Cache management, and Speculative Decoding. \n===============================================================\n連結智慧 守護世界 --- Connected Intelligence for Securing a Connected World","description_format":"text","description_chars":1822,"description_truncated":false,"requirements":{"experience_years_min":null,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Artificial Intelligence","Cybersecurity","Information Security","Email Security"],"lifecycle":[{"event":"open","at":"2026-09-29T07:28:08Z"}],"liveness":{"score":63,"band":"ok","label":"Likely open","p_open":1,"p_active":0.632,"p_room":1,"age_days":1,"expected_fill_days":50,"reasons":["conf:17","stale_co","velocity","win:early","comp:brand"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/trend-micro-staffsr-ml-infrastructure-platform-engineer","json_url":"https://alion.io/job/trend-micro-staffsr-ml-infrastructure-platform-engineer.json","meta":{"generated_at":"2026-10-01T12:51:16Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4374,"day_limit":5000,"remaining_today":626,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}