{"id":1458820,"url":"https://alion.io/job/infosys-spark-scala-databricks","title":"Spark-Scala, Databricks","company":{"id":223,"name":"Infosys","domain":"infosys.com","url":"https://alion.io/company/infosys","size_band":"5000+","is_staffing_agency":false,"employer_type":"services","is_intermediary":false,"listed_via":null,"ats_vendor":"Career site","truth_index":{"grade":"B","score":75,"open_postings":419,"ghost_share":0,"stale_share":1,"repost_share":0,"time_to_fill_p50_days":null,"computed_at":"2026-10-01T05:45:00Z"}},"role":"Data Science","role_family":"Data Science","seniority":"middle","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bengaluru, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":16000,"max_usd":39000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":13},"experience_years_min":3,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Databricks","optional":false},{"name":"ETL/ELT","optional":false},{"name":"Scala","optional":false},{"name":"Spark","optional":false},{"name":"Airflow","optional":true},{"name":"Azure","optional":true},{"name":"CI/CD","optional":true},{"name":"Delta Lake","optional":true},{"name":"Git","optional":true},{"name":"SQL","optional":true}],"status":"live","first_seen_at":"2026-09-19T08:03:51Z","employer_posted_date":"2026-09-29","last_verified_at":"2026-09-29T17:24:37Z","board_verified":true,"closed_at":null,"days_open":12,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":12},"description":"Join a collaborative engineering team where you’ll turn raw, complex data into trusted insights that power real business decisions. In this role, you’ll work hands-on with Spark-Scala and Databricks to build scalable data pipelines, optimize distributed processing, and help teams deliver reliable datasets for analytics and downstream applications. You’ll partner closely with data engineers, analysts, and stakeholders to understand requirements, translate them into robust ETL solutions, and continuously improve performance and quality. If you enjoy solving data challenges, learning modern cloud data practices, and taking ownership from development through production support, this is a great opportunity to grow your impact while working in a supportive, high-accountability culture.\nResponsibilities\nKey Responsibilities:\nData Engineering & ETL Development\nDesign, develop, and maintain ETL pipelines using Spark-Scala on Databricks for batch and incremental processing.\nImplement data transformations, joins, aggregations, and validations to ensure accurate and consistent outputs.\nBuild reusable Spark components and follow best practices for modular, maintainable code.\nPerformance, Reliability & Operations\nTune Spark jobs (partitioning, caching, shuffle optimization) to improve performance and cost efficiency.\nMonitor job execution, troubleshoot failures, and provide timely production support with root-cause analysis.\nImplement logging, error handling, and data quality checks to improve pipeline reliability.\nCollaboration & Delivery\nWork with cross-functional teams to gather requirements and translate them into technical solutions.\nParticipate in code reviews, documentation, and knowledge sharing to uplift team standards.\nSupport release cycles by validating outputs, ensuring backward compatibility, and maintaining deployment readiness.\nTechnical requirements\nPrimary skills:Technology->Big Data - Data Processing->Spark\nTechnology->Data Engineering->Databricks\nTechnology->Functional Programming->Scala\nAdditional responsibilities\nMinimum Qualifications:\nBachelor’s/Master’s degree (BE/BTech/MSc/MCA/MTech or equivalent).\n3-5 years of experience in data engineering or ETL development roles.\nStrong hands-on experience with Spark using Scala and working on Databricks.\nSolid understanding of ETL concepts, data transformations, and pipeline troubleshooting.\nAbility to write clean, testable code and collaborate effectively with technical and non-technical stakeholders.\nPreferred Qualifications:\nExperience building end-to-end pipelines on Databricks including notebooks, jobs/workflows, and cluster configuration basics.\nStrong SQL skills and experience integrating Spark pipelines with structured data sources and curated datasets.\nFamiliarity with data quality frameworks, reconciliation strategies, and automated validation checks.\nExposure to CI/CD practices for data engineering (version control, automated testing, release management).\nProven ability to optimize distributed workloads and deliver measurable improvements in runtime and stability.\nGood to have skills:\nSQL, Delta Lake, Apache Airflow, Azure Data Lake Storage (ADLS), Git\nPreferred skills\nTechnology->Big Data - Data Processing->Spark,Technology->Functional Programming->Scala,Technology->Data Engineering->Databricks\nEducation\nMCA,MSc,MTech,Bachelor of Engineering,BTech","description_format":"text","description_chars":3370,"description_truncated":false,"requirements":{"experience_years_min":3,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":true},"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["IT Consulting & Digital Transformation","IT Outsourcing & Dedicated Teams"],"lifecycle":[{"event":"open","at":"2026-09-29T11:08:16Z"}],"liveness":{"score":53,"band":"ok","label":"Likely open","p_open":1,"p_active":0.583,"p_room":0.9,"age_days":11,"expected_fill_days":18,"reasons":["conf:36","stale_co","velocity","win:mid","comp:brand"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/infosys-spark-scala-databricks","json_url":"https://alion.io/job/infosys-spark-scala-databricks.json","meta":{"generated_at":"2026-10-01T16:58:19Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":44,"day_limit":5000,"remaining_today":4956,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}