{"id":1235008,"url":"https://alion.io/job/sea-limited-llm-algorithm-engineer-post-training","title":"LLM Algorithm Engineer (Post Training)","company":{"id":159,"name":"Sea Limited","domain":"sea.com","url":"https://alion.io/company/sea-limited","size_band":"5000+","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Career site","truth_index":null},"role":"AI/ML","role_family":"AI/ML","seniority":null,"employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Singapore"],"countries":["SG"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":135000,"max_usd":294000,"period":"year","method":"role_country_seniority_unknown","sample_n":17},"experience_years_min":null,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"DPO","optional":false},{"name":"Function Calling","optional":false},{"name":"Hallucination","optional":false},{"name":"LLM","optional":false},{"name":"Megatron-LM","optional":false},{"name":"Post-training","optional":false},{"name":"PPO","optional":false},{"name":"Reward Modeling","optional":false},{"name":"RLHF","optional":false},{"name":"SFT","optional":false},{"name":"Synthetic Data","optional":false},{"name":"Tool Use","optional":false}],"status":"live","first_seen_at":"2026-09-25T16:14:42Z","employer_posted_date":"2026-09-25","last_verified_at":"2026-09-25T23:50:38Z","board_verified":true,"closed_at":null,"days_open":1,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":1},"description":"Post-Training Pipeline Implementation\nParticipate in the development and deployment of post-training pipelines such as SFT, DPO, PPO, and Reward Modeling; responsible for concrete execution from training, hyperparameter tuning, evaluation to online regression.\nParticipate in training optimization in Agent and Tool-use directions, improving metrics such as tool-calling accuracy, multi-turn instruction following, and hallucination control.\nParticipate in iteration of core-scenario models such as Router, Query rewriting, Agent tool calling, Web Search decision-making, Memory, and e-commerce search relevance.\nData and Evaluation\nManage the collection, cleaning, synthesis, and annotation workflows for post-training data; participate in annotation guideline development and vendor coordination.\nBuild and maintain automated evaluation combining human review and LLM-as-Judge; produce reproducible effectiveness reports.\nKeep up to date with frontier methods in the industry and experiment with them based on the team's direction.\nMaster's degree in Computer Science, AI, or a related field.\nPrior development experience in at least one LLM post-training area (SFT, DPO, PPO, RLHF, Reward Modeling, etc.); able to independently complete key steps from data through training to evaluation.\nPrior experience with at least one mainstream training framework (e.g., Megatron, veRL, etc.) and understanding of basic principles of distributed training.\nUnderstanding of Agent / Tool-use / multi-turn dialogue; familiar with data construction and basic alignment approaches for Function Calling.\nStrong engineering and troubleshooting capabilities; able to make reasonable trade-offs between effectiveness and iteration efficiency.\nPrior End-to-end post-training deployment experience, or participation in 1000-GPU-scale distributed training.\nPrior experience with leading large model teams will be a strong plus.\nA strong understanding or prior hands on experience in Agent RL, Self-play, synthetic data, or inference-time compute will be a strong plus","description_format":"text","description_chars":2048,"description_truncated":false,"requirements":{"experience_years_min":null,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"master","optional":false},"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Mobile Games","Marketplaces","Payment Processing & Gateways","Game Development"],"lifecycle":[{"event":"open","at":"2026-09-25T16:14:42Z"}],"liveness":{"score":86,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.86,"p_room":1,"age_days":0,"expected_fill_days":13,"reasons":["conf:5","win:early"],"computed_at":"2026-09-26T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/sea-limited-llm-algorithm-engineer-post-training","json_url":"https://alion.io/job/sea-limited-llm-algorithm-engineer-post-training.json","meta":{"generated_at":"2026-09-27T00:22:24Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":331,"day_limit":5000,"remaining_today":4669,"minute_limit":60,"resets_at":"2026-09-28T00:00:00Z"}}}