{"id":781567,"url":"https://alion.io/job/cardboard-senior-applied-ml-engineer-evals-data","title":"Senior Applied ML Engineer, Evals & Data","company":{"id":670001,"name":"Cardboard","domain":"cardboard.ai","url":"https://alion.io/company/cardboard-2","size_band":"51-200","is_staffing_agency":false,"is_intermediary":false,"ats_vendor":"Ashby","truth_index":null},"role":"AI/ML","role_family":"AI/ML","seniority":"senior","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bengaluru, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":40000,"max_usd":103000,"period":"year","method":"role_seniority_country_cell","sample_n":10},"experience_years_min":8,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AI Agents","optional":false},{"name":"Fine-tuning","optional":false},{"name":"LLM","optional":false},{"name":"Python","optional":false},{"name":"TypeScript","optional":false},{"name":"Multimodal AI","optional":true}],"status":"live","first_seen_at":"2026-08-15T11:14:50Z","employer_posted_date":"2026-08-15","last_verified_at":"2026-09-24T12:18:52Z","board_verified":true,"closed_at":null,"days_open":40,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":40},"description":"About\nWe're building the future of storytelling and video editing.\n\nWe're a small team that moves fast and builds things we're proud of.\n\nWe care obsessively about taste: in design, in product, in every detail.\n\nWe're solving these hard problems.\n\nWe're backed by a Tier-1 global fund, YC, and founders of billion dollar companies.\n\nEngineering\nVideo is the most powerful way humans tell stories. It always has been. But creating it today is still painfully hard. Fragmented tools, steep learning curves, and workflows that get in the way of the actual creative work. We're building Cardboard to change that.\nCardboard runs a real video editor in the browser, backed by a serious cloud media pipeline and an AI agent that actually understands footage. You will own how we measure and improve the quality of Cardboard’s AI agent.\nYou will study real agent runs, turn important failures into evaluation cases, and measure whether changes make the product better. You will also work with product and engineering to ship those improvements.\nThis is not a research-only, prompt-only, or QA role.\nYou’ll be working alongside a team of engineers who all care deeply about craft, including the founders. You like owning problems end to end, and you’d rather ship something great this week than something perfect next quarter\nWhat you'll actually do\nDefine quality standards and build trusted evaluation datasets from real product usage.\n\nBuild offline and online evaluations, including automated checks and human review.\n\nAnalyze model and agent failure patterns, then improve quality through better data, evaluation methods, model selection, and, where useful, fine-tuning.\n\nAdd regression checks and release gates while tracking quality, latency, and cost.\n\nSolve these hard problems.\n\nWhat we are looking for\nExperience shipping and operating an LLM or agent system used by real customers.\n\nStrong software engineering skills in TypeScript or Python, with the ability to work across both.\n\nExperience building evaluations, datasets, experiments, or AI quality systems.\n\nStrong product judgment and the ability to turn unclear quality problems into measurable improvements.\n\nYou do not need a PhD or experience training foundation models. Evidence of building reliable AI products matters more than formal credentials or knowledge of a specific framework.\nNice to have\nExperience with multimodal AI, video, media, or creative software.\n\nExperience with human labeling, model graders, or fine-tuning.\n\nGood knowledge of experiment design and statistics.\n\nWithin your first six months:\nWe have a trusted quality baseline for our main agent workflows.\n\nProduction failures regularly become new evaluation cases.\n\nImportant agent changes pass clear regression checks before release.\n\nWe can show measurable improvements in key editing workflows.\n\nWhat you get\nYou'd be surrounded by people who are absurdly good at what they do. One started coding at 11 and shipped an app with 6M+ downloads in high school. One got into CS engineering at 14 and has been working on distributed systems for 8+ years. One's an ex-founder who took a company to 1.2M users and $300M+ in transactions. That's the team. We're looking for someone who'll raise the bar on technical craftsmanship and creative product quality. Apart from that you'd get:\nCompetitive salary and founding-team equity.\n\nUnlimited tokens across every AI model. Use whatever you want, as much as you want.\n\nA healthy budget for AI tools and any peripherals you need to do your best work.","description_format":"text","description_chars":3530,"description_truncated":false,"requirements":{"experience_years_min":8,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"phd","optional":false},"security_clearance":false,"languages":[]},"benefits":["Equity"],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Manufacturing","Paper & Pulp Manufacturing","AI Agents"],"lifecycle":[{"event":"open","at":"2026-09-12T00:51:37Z"}],"liveness":{"score":49,"band":"ok","label":"Likely open","p_open":1,"p_active":0.659,"p_room":0.75,"age_days":39,"expected_fill_days":42,"reasons":["conf:4","win:late"],"computed_at":"2026-09-24T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/cardboard-senior-applied-ml-engineer-evals-data","json_url":"https://alion.io/job/cardboard-senior-applied-ml-engineer-evals-data.json","meta":{"generated_at":"2026-09-24T13:54:34Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers"}}