{"id":1315798,"url":"https://alion.io/job/toppan-merrill-senior-data-engineer","title":"Senior Data Engineer","company":{"id":2408462,"name":"Toppan Merrill","domain":"toppanmerrill.com","url":"https://alion.io/company/toppanmerrill","size_band":"501-1000","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Workday","truth_index":{"grade":"B","score":75,"open_postings":4,"ghost_share":0,"stale_share":1,"repost_share":0,"time_to_fill_p50_days":null,"computed_at":"2026-10-01T05:45:00Z"}},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Chennai, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":18000,"max_usd":37000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":51},"experience_years_min":null,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Azure","optional":false},{"name":"Azure Data Factory","optional":false},{"name":"CI/CD","optional":false},{"name":"Databricks","optional":false},{"name":"Delta Lake","optional":false},{"name":"ETL/ELT","optional":false},{"name":"IAM","optional":false},{"name":"pySpark","optional":false},{"name":"Python","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false},{"name":"Apache Kafka","optional":true},{"name":"dbt","optional":true},{"name":"MLFlow","optional":true}],"status":"live","first_seen_at":"2025-09-11T00:00:00Z","employer_posted_date":"2025-09-11","last_verified_at":"2026-10-01T03:56:14Z","board_verified":true,"closed_at":null,"days_open":385,"trust":{"level":"stale","repost_count":0,"flags":["stale"],"days_open":385},"description":"Job Description:\nResponsibilities:\nDevelop & Optimize Data PipelinesBuild, test, and maintainETL/ELT data pipelines using Azure Databricks & Apache Spark (PySpark).\nOptimizeperformance and cost-efficiency of Spark jobs.\nEnsure data quality through validation, monitoring, and alerting mechanisms.\nUnderstand cluster types, configuration, and use-case for serverless\n\nImplement Unity Catalog for Data GovernanceDesign and enforceaccess control policies using Unity Catalog.\nManagedata lineage, auditing, and metadata governance.\nEnable secure data sharing across teams and external stakeholders.\n\nIntegrate with Cloud Data PlatformsWork withAzure Data Lake Storage / Azure Blob Storage/ Azure Event Hub to integrate Databricks with cloud-baseddata lakes, data warehouses, and event streams.\nImplementDelta Lake for scalable, ACID-compliant storage.\n\nAutomate & Orchestrate WorkflowsDevelopCI/CD pipelines for data workflows usingAzureDatabricks Workflows or Azure Data Factory.\nMonitor and troubleshoot failures injob execution and cluster performance.\n\nCollaborate with StakeholdersWork withData Analysts, Scientists, and Business Teams to understand requirements.\nTranslate business needs intoscalable data engineering solutions.\n\nAPI expertiseAbility to pull data from a wide variety of APIs using different strategies and methods\n\nRequired Skills & Experience:\n Azure Databricks & Apache Spark (PySpark) - Strong experience in buildingdistributed data pipelines.\nPython - Proficiency in writing optimized and maintainable Python code for data engineering.\nUnity Catalog - Hands-on experience implementingdata governance, access controls, and lineage tracking.\nSQL - Strong knowledge of SQL for data transformations and optimizations.\nDelta Lake - Understanding oftime travel, schema evolution, and performance tuning.\nWorkflow Orchestration - Experience withAzureDatabricks Jobs or Azure Data Factory.\n CI/CD & Infrastructure as Code (IaC) - Familiarity withDatabricks CLI, Databricks DABs, and DevOps principles.\nSecurity & Compliance - Knowledge of IAM, role-based access control (RBAC), and encryption.\nPreferred Qualifications:\nExperience withMLflow for model tracking & deployment in Databricks.\nFamiliarity with streaming technologies (Kafka, Delta Live Tables, Azure Event Hub, Azure Event Grid).\nHands-on experience with dbt (Data Build Tool) for modular ETL development.\nCertification inDatabricks, Azure is a plus.\nExperience with Azure Databricks Lakehouse connectors for SalesForce and SQL Server\nExperience with Azure Synapse Link for Dynamics, dataverse\nFamiliarity with other data pipeline strategies, like Azure Functions, Fabric, ADF, etc\nSoft Skills:\nStrongproblem-solving and debugging skills.\nAbility towork independently and in teams.\nExcellent communication and documentation skills.","description_format":"text","description_chars":2808,"description_truncated":false,"requirements":{"experience_years_min":null,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Security Compliance","RegTech, AML & Compliance"],"lifecycle":[{"event":"open","at":"2026-09-26T18:33:11Z"}],"liveness":{"score":8,"band":"cold","label":"Long shot","p_open":1,"p_active":0.303,"p_room":0.28,"age_days":385,"expected_fill_days":39,"reasons":["conf:1","win:tail","crowd:"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/toppan-merrill-senior-data-engineer","json_url":"https://alion.io/job/toppan-merrill-senior-data-engineer.json","meta":{"generated_at":"2026-10-01T10:13:53Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1319,"day_limit":5000,"remaining_today":3681,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}