{"id":1242562,"url":"https://alion.io/job/kuailu-software-large-language-model-llm-evaluation-engineer","title":"Large Language Model (LLM) Evaluation Engineer","company":{"id":3802225,"name":"Kuailu Software","domain":"kuailutech.com","url":"https://alion.io/company/kuailu-software","size_band":"201-500","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"AI/ML","role_family":"AI/ML","seniority":"middle","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Singapore"],"countries":["SG"],"hiring_countries":[],"hiring_countries_total":0,"salary":{"min":6000,"max":12000,"currency":"SGD","period":"month","gross":true,"usd_annual":112752},"salary_estimate":null,"experience_years_min":3,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"LLM","optional":false},{"name":"lm-eval-harness","optional":false},{"name":"Pre-training","optional":false},{"name":"Python","optional":false},{"name":"SGLang","optional":false},{"name":"vLLM","optional":false},{"name":"CI/CD","optional":true}],"status":"live","first_seen_at":"2026-09-24T00:00:00Z","employer_posted_date":null,"last_verified_at":"2026-09-24T00:00:00Z","board_verified":false,"closed_at":null,"days_open":10,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":10},"description":"Job ResponsibilitiesBuild and maintain an automated LLM evaluation pipeline covering multiple dimensions, including general capabilities, Agent capabilities, and persona/role-playing. The pipeline should support one-click evaluation, historical result comparison, and regression testing.\nConduct general capability evaluations using benchmarks such as MMLU, C-Eval, HumanEval, GSM8K, MATH, and IFEval, including benchmark deployment, execution, and results analysis.\nConduct Agent capability evaluations, including setting up evaluation environments and tracking metrics for benchmarks such as BFCL, τ-bench, and GAIA.\nDesign and execute persona/role-playing evaluation frameworks, covering metrics such as identity recognition, role compatibility, multi-turn stability, and style consistency.\nRecord and analyze evaluation results from training runs, conduct comparative analysis and anomaly detection, and produce checkpoint evaluation reports.\nConduct regular intermediate evaluations during the pre-training stage to track the evolution and improvement of model capabilities.\nJob RequirementsBachelor's degree or above in Computer Science, Artificial Intelligence, or a related field.\nFamiliarity with mainstream LLM evaluation benchmarks and frameworks, such as lm-eval-harness, OpenCompass, and EvalPlus.\nStrong proficiency in Python, with the ability to independently build evaluation pipelines covering model inference/deployment, batch evaluation, and results analysis.\nFamiliarity with LLM inference frameworks such as vLLM and SGLang, with the ability to deploy models for batch inference and evaluation.\nExperience in evaluation data analysis and visualization.\nDetail-oriented and rigorous, with a strong focus on ensuring the reproducibility and reliability of evaluation results.\nPreferred QualificationsExperience with Agent evaluation, particularly BFCL, τ-bench, GAIA, or SWE-bench.\nExperience with persona or role-playing evaluation, such as CharacterBench or RMTBench.\nExperience with evaluation automation and CI/CD integration.\nUnderstanding of model training workflows, with the ability to understand the relationship between training checkpoints and evaluation results.\nSkills\nTesting Results, Pipeline Management, Model Deployment, Mathematics, Pipeline Development, Artificial Intelligence, Computer Science, Data Evaluation, Capability Development, Reproducibility, Calculation Agent, Role Playing, Capability Analysis, Benchmarking, Checkpoint, Performance Evaluations","description_format":"text","description_chars":2496,"description_truncated":false,"requirements":{"experience_years_min":3,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-09-25T16:15:06Z"}],"visa":[],"liveness":{"score":90,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.902,"p_room":1,"age_days":9,"expected_fill_days":39,"reasons":["seen:9","velocity","win:early"],"computed_at":"2026-10-03T05:45:00Z"},"pay":{"stated_usd_annual":112752,"is_top_pay":false},"html_url":"https://alion.io/job/kuailu-software-large-language-model-llm-evaluation-engineer","json_url":"https://alion.io/job/kuailu-software-large-language-model-llm-evaluation-engineer.json","meta":{"generated_at":"2026-10-04T00:53:18Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1085,"day_limit":5000,"remaining_today":3915,"minute_limit":60,"resets_at":"2026-10-05T00:00:00Z"}}}