{"id":48520,"url":"https://alion.io/job/m-league-data-engineer","title":"Data Engineer","company":{"id":9447,"name":"M-League","domain":"mleague.gg","url":"https://alion.io/company/m-league","size_band":null,"is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"junior","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bengaluru, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":14000,"max_usd":33000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":11},"experience_years_min":2,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Apache Kafka","optional":false},{"name":"Claude","optional":false},{"name":"Claude Code","optional":false},{"name":"Copilot","optional":false},{"name":"Cursor","optional":false},{"name":"Databricks","optional":false},{"name":"ETL/ELT","optional":false},{"name":"GCP","optional":false},{"name":"Gemini","optional":false},{"name":"Google BigQuery","optional":false},{"name":"Google Cloud Run","optional":false},{"name":"LLM","optional":false},{"name":"Model Context Protocol","optional":false},{"name":"pySpark","optional":false},{"name":"Python","optional":false},{"name":"Snowflake","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false}],"status":"live","first_seen_at":"2026-08-10T05:31:58Z","employer_posted_date":null,"last_verified_at":"2026-08-10T05:31:58Z","board_verified":false,"closed_at":null,"days_open":56,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":56},"description":"The candidate will have responsibilities across the following functions:\nData Warehousing and Modelling:\nOwn the design of silver/gold layer tables: grain, join keys, schema, partitioning, and clustering decisions for large-scale datasets.\nBuild and maintain batch and streaming ETL pipelines (using big data technologies) feeding the medallion architecture.\nDesign idempotent, observable pipelines with QC gates, backfill strategies, and clear failure semantics.\nStandardise data models and conventions across multiple game products/projects.\nDeliver high-quality data pipelines with utmost importance to correctness and availability of the data to different target audiences/systems.\nDesign Decisions and Technical Ownership:\nLead source-system profiling and drive grain, deduplication, and enrichment decisions backed by data, not assumptions.\nOwn trade-off calls: cost vs. freshness, load strategy, schema evolution, and query/slot optimisation.\nDocument designs and decisions so they survive beyond the author.\nStakeholder Management:\nPartner with analysts, PMs, and game teams to translate business questions into warehouse requirements and reconcile instrumentation gaps.\nCommunicate design proposals and data profiling findings to both technical and leadership audiences.\nRequirements:\nExperience: 2+ years in data engineering, with deep data warehousing exposure at scale.\nSQL and BigQuery: Expert-level SQL; hands-on BigQuery optimisation (partitioning, clustering, slots, cost).\nData Modelling: Strong command of medallion architecture, dimensional/fact modelling, grain definition, and schema design.\nProgramming: Production-grade Python; PySpark on Dataproc (or Databricks/EMR equivalents).\nStreaming: Working experience with Kafka-based ingestion (Confluent preferred).\nData Quality: Experience building validation, observability, and reconciliation into pipelines.\nStakeholder Skills: Proven ability to gather requirements, challenge assumptions, and communicate trade-offs clearly.\nEducation: Bachelor's/Master's in CS, Engineering, or related field (or equivalent experience).\nAI-Augmented Engineering (Must-Have):\nWe expect AI tooling to be part of your daily workflow, with sound judgment on when to trust and when to verify:\nHands-on daily use of AI coding agents Claude Code / CLI, Claude web app, or equivalents (Cursor, Copilot, Gemini CLI).\nExperience creating reusable AI assets: custom skills/commands, prompt libraries, or repo-level agent context.\nFamiliarity with MCP or similar integrations connecting agents to warehouses and internal tools.\nAbility to demonstrate real examples of AI-accelerated work and where you drew the human-review line.\nPreferred:\nLLM-in-pipeline experience: AI-driven data quality checks, anomaly triage, or text-to-SQL for self-service analytics.\nBroader GCP: Pub/Sub, Cloud Run, Cloud Functions, Cloud SQL.\nGaming, fintech, or other high-volume consumer event domains.","description_format":"text","description_chars":2926,"description_truncated":false,"requirements":{"experience_years_min":2,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":false},"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Gaming","Mobile Games","Game Development"],"lifecycle":[{"event":"open","at":"2026-08-10T05:31:58Z"}],"visa":[],"liveness":{"score":4,"band":"cold","label":"Long shot","p_open":0.4,"p_active":0.356,"p_room":0.28,"age_days":56,"expected_fill_days":16,"reasons":["seen:56","win:tail","crowd:junior"],"computed_at":"2026-10-05T05:45:15Z"},"pay":null,"html_url":"https://alion.io/job/m-league-data-engineer","json_url":"https://alion.io/job/m-league-data-engineer.json","meta":{"generated_at":"2026-10-06T02:41:17Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","about":"Alion is a live layer of people, companies and AI agents: who they are, whether they are real and active right now, what they do and how to work with them, readable by people and by agents and paid per call.","catalog":"https://alion.io/catalog.json","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4056,"day_limit":5000,"remaining_today":944,"minute_limit":60,"resets_at":"2026-10-07T00:00:00Z"}},"offers":[{"id":"company.slices","title":"One company in depth, by slice","status":"live","price":{"credits":0.02,"usd":0.002,"plus_per_slice":{"credits":0.05,"usd":0.005}},"unit":"per company, plus each slice with data","note":"the employer in depth","call":{"mcp_tool":"get_company","arguments":{"id":9447},"rest":"https://alion.io/mcp/rest/get_company?id=9447"},"human":"https://alion.io/catalog?offer=company.slices&for=job%2Fm-league-data-engineer"},{"id":"market.stats","title":"A market slice: pay, demand and time to fill","status":"live","price":{"credits":1,"usd":0.1},"unit":"per slice","note":"pay, demand and time to fill for this role and place","call":{"mcp_tool":"market_stats"},"human":"https://alion.io/catalog?offer=market.stats&for=job%2Fm-league-data-engineer"},{"id":"job.search","title":"Open jobs by role, technology, place, pay and visa","status":"live","price":{"credits":0.02,"usd":0.002},"unit":"per posting in a list","note":"similar open postings","call":{"mcp_tool":"search_jobs"},"human":"https://alion.io/catalog?offer=job.search&for=job%2Fm-league-data-engineer"},{"id":"company.verify","title":"Is this company real and active right now","status":"pilot","price":null,"unit":"per company","request":{"url":"https://alion.io/catalog/request","method":"POST","body":"{\"offer\": \"company.verify\", \"for\": \"job/m-league-data-engineer\", \"note\": \"what you need it for\"}"},"human":"https://alion.io/catalog?offer=company.verify&for=job%2Fm-league-data-engineer"}]}