{"id":1194601,"url":"https://alion.io/job/clera-data-scientist-agent-evaluations-quality-6","title":"Data Scientist, Agent Evaluations & Quality","company":{"id":2706,"name":"Clera","domain":"getclera.com","url":"https://alion.io/company/clera","size_band":"11-50","is_staffing_agency":true,"employer_type":"agency","is_intermediary":false,"listed_via":null,"ats_vendor":"Ashby","truth_index":{"grade":"B","score":80,"open_postings":30,"ghost_share":0,"stale_share":1,"repost_share":0,"time_to_fill_p50_days":6,"computed_at":"2026-09-26T05:45:00Z"}},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":"full_time","work_mode":"remote","remote_scope":"stated_countries","remote_scope_basis":"posting_text","remote_working_hours":null,"hiring_geo_confidence":"inferred","locations":["Palo Alto, United States"],"countries":["US"],"hiring_countries":["US"],"hiring_countries_total":1,"salary":null,"salary_estimate":{"min_usd":141000,"max_usd":242000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":184},"experience_years_min":5,"visa_sponsorship":true,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AI Agents","optional":false},{"name":"Function Calling","optional":false},{"name":"LLM","optional":false},{"name":"Machine Learning","optional":false},{"name":"Python","optional":false},{"name":"SQL","optional":false},{"name":"Tool Use","optional":false}],"status":"live","first_seen_at":"2026-09-24T17:09:53Z","employer_posted_date":"2026-09-24","last_verified_at":"2026-09-27T05:17:20Z","board_verified":true,"closed_at":null,"days_open":2,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":2},"description":"About the Role\nThis role sits at the intersection of applied data science and AI product quality for a small, fast-moving AI productivity startup building autonomous agents that handle email, calendar, browser, and business software tasks. You will own the measurement of agent quality end-to-end: turning ambiguous product behavior into rigorous, actionable evaluation systems that directly guide engineering and product decisions.\nWhat You'll Do\nArchitect and maintain automated evaluation pipelines that measure agent quality across capabilities and product surfaces.\n\nTranslate agent capabilities into explicit success criteria, including pass, partial-pass, and failure definitions for complex multi-step tasks.\n\nBuild representative gold datasets and regression suites covering common workflows, edge cases, ambiguous requests, and adversarial scenarios.\n\nDefine and track metrics such as task success, tool-selection accuracy, instruction adherence, factual consistency, latency, cost, and reliability.\n\nDesign deterministic and model-based graders, calibrate LLM-as-a-judge systems, and measure grader agreement, false positives, and false negatives.\n\nAnalyze traces, tool calls, model outputs, and production outcomes to identify root causes and build a useful failure taxonomy.\n\nCompare models, prompts, tools, and capability implementations using rigorous offline experiments and production evidence.\n\nBuild dashboards and release-quality signals that make evaluation results understandable and actionable for engineering, product, and leadership.\n\nPartner with capability engineers to recommend improvements and verify that fixes raise quality without unacceptable regressions in cost, latency, or reliability.\n\nWhat We're Looking For\n5+ years in data science, machine learning, or analytics roles, with a focus on evaluation systems, metrics frameworks, or quality measurement for production systems.\n\nDemonstrated experience designing and implementing evaluation frameworks, grading systems, and success criteria for ML or AI systems in production.\n\nStrong Python and SQL proficiency with the ability to build automated data pipelines and production-quality analysis code at scale.\n\nSolid statistical and experimental design knowledge: sampling, variance, uncertainty quantification, bias detection, confounding variables, and significance testing for non-deterministic systems.\n\nExperience with ground-truth data development: labeling guideline design, annotation quality control, ambiguity resolution, and dataset maintenance.\n\nWorking knowledge of LLM behavior, tool use, retrieval systems, multi-step execution, and practical failure modes of language model systems.\n\nAbility to connect quantitative patterns to individual system traces and identify failure origins across model, prompt, context, tools, data, and application logic.\n\nExperience communicating evaluation results, methodology, uncertainty, and trade-offs to both technical and non-technical stakeholders.\n\nComfort operating with high ownership in ambiguous, fast-moving environments, independently turning open-ended quality questions into evaluation systems.\n\nExperience with LLM-as-a-judge systems, agentic or multi-step task evaluation, or benchmarking platforms for AI systems is a strong plus.\n\nLocation\nOn-site in Palo Alto, California, United States. Visa sponsorship is not available for this role.","description_format":"text","description_chars":3389,"description_truncated":false,"requirements":{"experience_years_min":5,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[{"name":"United States","iso":"US","kind":"country"},{"name":"Palo Alto","iso":null,"kind":"city"}],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-09-24T18:32:37Z"}],"liveness":{"score":38,"band":"fade","label":"Fading","p_open":1,"p_active":0.379,"p_room":1,"age_days":1,"expected_fill_days":6,"reasons":["conf:2","agency","stale_co","velocity","win:early","comp:brand"],"computed_at":"2026-09-26T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/clera-data-scientist-agent-evaluations-quality-6","json_url":"https://alion.io/job/clera-data-scientist-agent-evaluations-quality-6.json","meta":{"generated_at":"2026-09-27T05:29:01Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4795,"day_limit":5000,"remaining_today":205,"minute_limit":60,"resets_at":"2026-09-28T00:00:00Z"}}}