{"id":1230351,"url":"https://alion.io/job/neemtree-data-engineer","title":"Data Engineer","company":{"id":3800233,"name":"Neemtree","domain":"neemtree.in","url":"https://alion.io/company/neemtree-3","size_band":null,"is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"middle","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Mumbai, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":21000,"max_usd":53000,"period":"year","method":"global_role_cell_scaled_by_country","sample_n":431},"experience_years_min":3,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Airflow","optional":false},{"name":"Amazon Kinesis","optional":false},{"name":"Amazon Redshift","optional":false},{"name":"Amazon S3","optional":false},{"name":"Apache Kafka","optional":false},{"name":"AWS","optional":false},{"name":"AWS Step Functions","optional":false},{"name":"CI/CD","optional":false},{"name":"Dimensional Modeling","optional":false},{"name":"ETL/ELT","optional":false},{"name":"Flink","optional":false},{"name":"Git","optional":false},{"name":"Linux","optional":false},{"name":"Python","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false},{"name":"Unix","optional":false},{"name":"Amazon ECS","optional":true},{"name":"Amazon EKS","optional":true},{"name":"Apache Hudi","optional":true},{"name":"Apache Iceberg","optional":true},{"name":"AWS Lambda","optional":true},{"name":"Delta Lake","optional":true},{"name":"Docker","optional":true},{"name":"IAM","optional":true},{"name":"Kubernetes","optional":true},{"name":"pySpark","optional":true}],"status":"live","first_seen_at":"2026-09-19T05:50:01Z","employer_posted_date":null,"last_verified_at":"2026-09-19T05:50:01Z","board_verified":false,"closed_at":null,"days_open":8,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":8},"description":"About the Role : \n\nWe're seeking a Data Engineer with 3 - 4 years of experience to join our growing tech team. The ideal candidate will have hands-on experience in building and managing scalable data systems on AWS using Spark and modern data frameworks.\n\nKey Responsibilities : \n\n- Design, build, and maintain end-to-end data pipelines for ingestion, transformation, and delivery of high-volume data.\n\n- Develop Spark-based ETL/ELT workflows for both batch and real-time streaming data.\n\n- Integrate data from multiple internal and external systems using Kafka, Kinesis, or other streaming frameworks.\n\n- Build and manage data models, warehouses, and lakehouses using AWS services such as S3, Glue, Redshift, Athena, etc.\n\n- Implement data quality checks, validation rules, and monitoring to ensure reliability and consistency.\n\n- Collaborate with data analysts and scientists to provide clean, structured datasets optimized for analytics and ML.\n\n- Work with orchestration tools (Airflow, MWAA, Step Functions, etc.) for automated workflow scheduling.\n\n- Continuously optimize data pipelines for cost, scalability, and performance.\n\nTech Stack : \n\n- Spark, AWS (S3, Glue, Redshift, Kinesis), Kafka, Airflow, Python, SQL.\n\nRequired Skills : \n\n- Strong programming skills in Python for data manipulation and automation.\n\n- Hands-on expertise in Apache Spark (PySpark or Spark SQL) for large-scale data processing.\n\n- Deep understanding of AWS data ecosystem - S3, Glue, Lambda, Redshift, Athena, EMR, Kinesis, IAM.\n\n- Experience with real-time streaming platforms such as Kafka, Kinesis, or Flink.\n\n- Strong command of SQL and data modeling (star schema, dimensional modeling, partitioning).\n\n- Proficiency with data orchestration and workflow management tools (Airflow, Step Functions, etc.).\n\n- Familiarity with Git, CI/CD, and modern development best practices.\n\n- Experience working in Linux/Unix environments and handling large datasets efficiently.\n\nGood to Have : \n\n- Exposure to data lakehouse technologies (Delta Lake, Iceberg, Hudi).\n\n- Understanding of data governance, cataloging, and lineage tools (Glue Data Catalog, Amundsen, DataHub).\n\n- Familiarity with containerization and deployment (Docker, ECS, EKS).\n\n- Basic understanding of AI/ML.\n\nQualifications : \n\n- Bachelor's degree in Computer Science, IT, Engineering, or related technical field.\n\n- 3 - 4 years of experience in data engineering, big data, or analytics infrastructure.\n\nSkills\nData Engineering, AWS, Data Ingestion, ETL, Kafka, Spark, DataLake, Data Modeling, Data Warehousing, Python, SQL","description_format":"text","description_chars":2571,"description_truncated":false,"requirements":{"experience_years_min":3,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Professional Services","Recruiting & Staffing"],"lifecycle":[{"event":"open","at":"2026-09-25T14:00:00Z"}],"liveness":{"score":79,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.835,"p_room":0.945,"age_days":7,"expected_fill_days":17,"reasons":["seen:7","velocity","win:mid"],"computed_at":"2026-09-27T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/neemtree-data-engineer","json_url":"https://alion.io/job/neemtree-data-engineer.json","meta":{"generated_at":"2026-09-28T03:04:38Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1933,"day_limit":5000,"remaining_today":3067,"minute_limit":60,"resets_at":"2026-09-29T00:00:00Z"}}}