{"id":1227553,"url":"https://alion.io/job/ekfrazo-technologies-private-limited-senior-data-engineer","title":"Senior Data Engineer","company":{"id":3800102,"name":"Ekfrazo Technologies","domain":"ekfrazo.in","url":"https://alion.io/company/ekfrazo-technologies-private-limited","size_band":"1001-5000","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":null,"work_mode":"remote","remote_scope":"stated_countries","remote_scope_basis":"board_field","remote_working_hours":null,"hiring_geo_confidence":"inferred","locations":["India"],"countries":["IN"],"hiring_countries":["IN"],"hiring_countries_total":1,"salary":null,"salary_estimate":{"min_usd":21000,"max_usd":42000,"period":"year","method":"role_seniority_country_cell","sample_n":57},"experience_years_min":8,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AI Agents","optional":false},{"name":"Azure","optional":false},{"name":"Azure AKS","optional":false},{"name":"Azure Data Factory","optional":false},{"name":"Azure DevOps","optional":false},{"name":"Bicep","optional":false},{"name":"CI/CD","optional":false},{"name":"Databricks","optional":false},{"name":"Delta Lake","optional":false},{"name":"Dimensional Modeling","optional":false},{"name":"Docker","optional":false},{"name":"Embeddings","optional":false},{"name":"ETL/ELT","optional":false},{"name":"FastAPI","optional":false},{"name":"Git","optional":false},{"name":"Kubernetes","optional":false},{"name":"Microsoft Defender","optional":false},{"name":"Microsoft Defender for Cloud","optional":false},{"name":"Microsoft Entra ID","optional":false},{"name":"Microsoft Fabric","optional":false},{"name":"MS SQL","optional":false},{"name":"Power BI","optional":false},{"name":"pySpark","optional":false},{"name":"Python","optional":false},{"name":"RAG","optional":false},{"name":"Rest API","optional":false},{"name":"SLI/SLO/SLA","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false},{"name":"Terraform","optional":false},{"name":"Trivy","optional":false},{"name":"Grafana","optional":true},{"name":"LangChain","optional":true},{"name":"LangGraph","optional":true},{"name":"MLFlow","optional":true},{"name":"Model Context Protocol","optional":true},{"name":"OpenAI","optional":true},{"name":"Prometheus","optional":true},{"name":"SSIS","optional":true}],"status":"live","first_seen_at":"2026-09-24T04:22:51Z","employer_posted_date":null,"last_verified_at":"2026-09-24T04:22:51Z","board_verified":false,"closed_at":null,"days_open":7,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":7},"description":"Job Description:\n\nAbout the Role:\n\nWe are looking for a Senior Data Engineer to design, build, and operate enterprise-scale data and AI platforms on Microsoft Azure.\n\nYou will own the lakehouse and ETL/ELT layer end to end using Microsoft Fabric, Azure Databricks, and Azure Data Factory.\n\nYou will also package and ship data services, pipelines, and AI workloads as production-grade Docker containers.\n\nThe role suits someone who combines strong data engineering fundamentals with a DevOps mindset and is comfortable extending into Generative AI and RAG solutions.\n\nKey Responsibilities:\n\n- Design and deliver scalable, metadata-driven ETL/ELT pipelines using Azure Data Factory, Fabric Data Factory, Databricks Workflows, and Lakeflow Jobs.\n\n- Build and maintain lakehouse platforms using Medallion Architecture (Bronze, Silver, Gold) on OneLake, ADLS Gen2, and Delta Lake.\n\n- Develop high-quality PySpark, Spark SQL, Python, and T-SQL transformations covering incremental loads, CDC, data-quality rules, and business-rule validation.\n\n- Optimize Spark and Delta Lake workloads using AQE, partitioning, join strategies, Z-ORDER, Liquid Clustering, data skipping, and file compaction.\n\n- Containerize data applications, ingestion services, APIs, and pipeline components using Docker.\n\n- Own the full container lifecycle: build, tag, scan, publish, deploy, and monitor.\n\n- Author optimized, secure multi-stage Dockerfiles and docker-compose setups for local development, testing, and integration.\n\n- Manage container images in Azure Container Registry (ACR) and deploy to Azure Kubernetes Service (AKS) or Azure Container Apps.\n\n- Build CI/CD pipelines in Azure DevOps for automated container builds, testing, vulnerability scanning, and release.\n\n- Use Fabric Deployment Pipelines and Databricks Asset Bundles for platform deployments.\n\n- Implement data governance and security using Unity Catalog, Microsoft Entra ID, RBAC, Managed Identity, and Azure Key Vault.\n\n- Build RAG pipelines and AI Agent workflows (document parsing, chunking, embeddings, vector search, model serving), and deploy agent and retrieval services as containerized microservices.\n\n- Enable self-service analytics through SQL Endpoints, Power BI datasets, and natural-language interfaces such as Databricks Genie.\n\n- Own production support: SLA monitoring, incident troubleshooting, root-cause analysis, and performance tuning.\n\n- Mentor junior engineers, conduct code reviews, and contribute to engineering standards and best practices.\n\nRequired Skills & Experience:\n\nData Engineering Core:\n\n- 8+ years in data engineering, ETL/ELT, and data warehousing.\n\n- Strong hands-on experience with Azure Databricks (Delta Lake, Auto Loader, Unity Catalog, Lakeflow/DLT, Databricks Workflows).\n\n- Strong hands-on experience with Microsoft Fabric (OneLake, Lakehouse, Data Warehouse, Fabric Data Factory, Notebooks, Dataflows Gen2, SQL Endpoint).\n\n- Expert in Azure Data Factory, Azure Synapse Analytics, and ADLS Gen2.\n\n- Advanced proficiency in Python, PySpark, Spark SQL, and SQL/T-SQL, including stored procedures, query tuning, and performance optimization.\n\n- Proven experience with Spark and Delta Lake performance tuning at scale.\n\n- Solid grounding in dimensional modelling and data-quality frameworks.\n\nDocker & Containerization (Strong, Hands-on):\n\n- 3+ years of production experience with Docker, covering image design, layering, caching, and size and security optimization.\n\n- Proficient in writing multi-stage Dockerfiles, docker-compose, and container networking, volumes, and environment/secret management.\n\n- Experience containerizing Python and PySpark applications, data ingestion services, REST/FastAPI services, and AI/RAG microservices.\n\n- Hands-on experience with Azure Container Registry (ACR), plus deployment on AKS or Azure Container Apps.\n\n- Working knowledge of Kubernetes fundamentals (pods, deployments, services, config maps, secrets, scaling).\n\n- Container security practices: base-image hardening, non-root users, image scanning (e.g., Trivy or Microsoft Defender for Cloud), and dependency management.\n\n- Ability to integrate container builds and deployments into CI/CD pipelines.\n\n- Experience with Databricks custom containers or containerized job runtimes is a strong plus.\n\nDevOps & Governance:\n\n- Azure DevOps, Git, and CI/CD pipelines for data and container workloads.\n\n- Infrastructure-as-code exposure (Terraform or Bicep), Databricks Asset Bundles, and Fabric Deployment Pipelines.\n\n- Unity Catalog, Entra ID, RBAC, Managed Identity, and Key Vault.\n\nGood to Have:\n\n- Generative AI and Agentic AI: RAG, Azure OpenAI / Azure AI Foundry, Azure AI Search, Databricks Mosaic AI Vector Search and Model Serving, LangChain/LangGraph, MLflow, and Model Context Protocol (MCP).\n\n- Power BI and dashboard development.\n\n- Experience in BFSI or financial-services data platforms.\n\n- Exposure to SSIS/SSRS and legacy-to-cloud migration.\n\n- Observability tooling for containers and pipelines (Azure Monitor, Log Analytics, Prometheus/Grafana).\n\nCertifications (Preferred):\n\n- Microsoft Certified: Fabric Data Engineer Associate (DP-700)\n\n- Any of: Azure Data Engineer Associate (DP-203), Databricks Data Engineer Professional, Azure Developer (AZ-204), Certified Kubernetes Application Developer (CKAD)\n\nEducation:\n\nBachelor's degree in Engineering, Computer Science, or a related field.\n\nSoft Skills:\n\n- Strong ownership and end-to-end accountability for production platforms\n\n- Clear communication with business and technical stakeholders\n\n- Structured problem-solving and troubleshooting\n\n- Mentoring and collaborative working style\nSkills\nAzure Databricks, Microsoft Fabric, Azure Data Factory, Python, PySpark, Docker, Kubernetes, Delta Lake, SQL, Generative AI, Data Warehousing","description_format":"text","description_chars":5780,"description_truncated":false,"requirements":{"experience_years_min":8,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[{"name":"India","iso":"IN","kind":"country"}],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-09-25T13:06:44Z"}],"liveness":{"score":83,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.875,"p_room":0.945,"age_days":7,"expected_fill_days":20,"reasons":["seen:7","velocity","win:mid"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/ekfrazo-technologies-private-limited-senior-data-engineer","json_url":"https://alion.io/job/ekfrazo-technologies-private-limited-senior-data-engineer.json","meta":{"generated_at":"2026-10-01T19:43:20Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":2376,"day_limit":5000,"remaining_today":2624,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}