{"id":1252967,"url":"https://alion.io/job/clover-infotech-pyspark-data-engineer","title":"Pyspark Data Engineer","company":{"id":3806704,"name":"Clover Infotech","domain":"cloverinfotech.com","url":"https://alion.io/company/clover-infotech","size_band":null,"is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Chennai, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":20000,"max_usd":42000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":51},"experience_years_min":6,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Airflow","optional":false},{"name":"ETL/ELT","optional":false},{"name":"pySpark","optional":false},{"name":"Python","optional":true},{"name":"Spark","optional":true}],"status":"live","first_seen_at":"2026-09-11T11:02:49Z","employer_posted_date":null,"last_verified_at":"2026-09-11T11:02:49Z","board_verified":false,"closed_at":null,"days_open":20,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":20},"description":"Job Description : \n\nWe are seeking a highly skilled Data Engineer with deep expertise in PySpark and the Cloudera Data Platform (CDP) to join our data engineering team. As a Data Engineer, you will be responsible for designing, developing, and maintaining scalable data pipelines that ensure high data quality and availability across the organization.\n\nResponsibilities : \n\n- Data Pipeline Development : Design, develop, and maintain highly scalable and optimized ETL pipelines using PySpark on the Cloudera Data Platform, ensuring data integrity and accuracy.\n\n- Data Ingestion : Implement and manage data ingestion processes from a variety of sources (e.g., relational databases, APIs, file systems) to the data lake or data warehouse on CDP.\n\n- Data Transformation and Processing : Use PySpark to process, cleanse, and transform large datasets into meaningful formats that support analytical needs and business requirements.\n\n- Performance Optimization : Conduct performance tuning of PySpark code and Cloudera components, optimizing resource utilization and reducing runtime of ETL processes.\n\n- Data Quality and Validation : Implement data quality checks, monitoring, and validation routines to ensure data accuracy and reliability throughout the pipeline.\n\n- Automation and Orchestration : Automate data workflows using tools like Apache Oozie, Airflow, or similar orchestration tools within the Cloudera ecosystem.\n\n- Monitoring and Maintenance : Monitor pipeline performance, troubleshoot issues.\n\nSkills\nData Engineering, PySpark, ETL, Data Integrity, Data Ingestion, Data Warehousing, DataLake, Data Quality, Apache Airflow","description_format":"text","description_chars":1633,"description_truncated":false,"requirements":{"experience_years_min":6,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-09-25T18:04:06Z"}],"liveness":{"score":43,"band":"fade","label":"Fading","p_open":0.85,"p_active":0.674,"p_room":0.75,"age_days":19,"expected_fill_days":24,"reasons":["seen:19","win:late"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/clover-infotech-pyspark-data-engineer","json_url":"https://alion.io/job/clover-infotech-pyspark-data-engineer.json","meta":{"generated_at":"2026-10-01T21:02:03Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":3502,"day_limit":5000,"remaining_today":1498,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}