{"id":1571799,"url":"https://alion.io/job/optiveum-senior-data-engineer-databricks-pyspark","title":"Senior Data Engineer (Databricks & PySpark)","company":{"id":685848,"name":"Optiveum","domain":"optiveum.com","url":"https://alion.io/company/optiveum","size_band":null,"is_staffing_agency":true,"employer_type":"agency","is_intermediary":false,"listed_via":null,"ats_vendor":"Teamtailor","truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":"full_time","work_mode":"remote","remote_scope":"stated_countries","remote_scope_basis":"inferred_company_offices","remote_working_hours":null,"hiring_geo_confidence":"inferred","locations":["Warsaw, Poland"],"countries":["PL"],"hiring_countries":["PL"],"hiring_countries_total":1,"salary":{"min":null,"max":38,"currency":"EUR","period":"hour","gross":null,"usd_annual":86000},"salary_estimate":null,"experience_years_min":5,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Azure","optional":false},{"name":"CI/CD","optional":false},{"name":"Databricks","optional":false},{"name":"Delta Lake","optional":false},{"name":"Git","optional":false},{"name":"Platform Engineering","optional":false},{"name":"pySpark","optional":false},{"name":"Python","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false}],"status":"live","first_seen_at":"2026-09-29T08:38:28Z","employer_posted_date":"2026-09-29","last_verified_at":"2026-10-03T23:19:37Z","board_verified":true,"closed_at":null,"days_open":4,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":4},"description":"Senior Data Engineer (Databricks & PySpark)\nLocation: 100% Remote (Warsaw, PL)\nRate: Up to 38 EUR / hour\nMode of work: Full-time (40 hours/week)\nContracting party: Optiveum (B2B Cooperation Agreement)\nAbout the role:\nWe are seeking an experienced Senior Data Engineer on behalf of our client to design, build, and operate their Sustainability Data Foundation (SDF). The SDF supports trusted, scalable, and auditable sustainability reporting and analytics through a modern data platform built on Databricks.\nYou will join a Sustainability Data & AI team that values reliability, operational excellence, and practical solutions over unnecessary complexity. The focus of this role is to deliver data products and pipelines that enable regulatory compliance and advanced analytics.\nKey Responsibilities:\nData Engineering & Product Development: Design and implement scalable data products using Databricks, Delta Lake, and PySpark. Build and maintain data pipelines processing complex datasets from ERP, procurement, sustainability, and external systems.\n\nPerformance Optimization: Optimize workloads for performance, scalability, and cost efficiency, building reusable engineering patterns.\n\nData Quality & Governance: Implement automated data quality controls throughout the data lifecycle. Proactively identify and resolve data issues before they impact reporting.\n\nPlatform Engineering & DevOps: Implement CI/CD pipelines, automated deployments, testing frameworks, and Infrastructure as Code (IaC). Support platform security controls and access management.\n\nStakeholder Collaboration: Partner with sustainability experts, business analysts, and reporting teams to support requirements gathering, solution design, and production releases.\n\nRequired Qualifications:\n5-7 years of experience designing, developing, and operating data platforms and pipelines.\n\nExtensive hands-on experience with Databricks (including Azure Databricks and Databricks Workflows) in production environments.\n\nExpert-level proficiency in PySpark, Python, Spark SQL, Data Modelling, and Data Pipeline Design.\n\nExperience implementing CI/CD pipelines for data engineering workloads, Git-based development, and version control.\n\nSolid understanding of data lineage, metadata management, governance, auditing, and validation frameworks.\n\nFluent English (both written and spoken).\n\nProactive, self-driven mindset with strong troubleshooting and root-cause analysis skills.\n\nNice to have:\nExperience with sustainability reporting, ESG data, and regulatory reporting requirements.\n\nKnowledge of procurement, supplier, finance, and ERP data domains.\n\nExperience developing Databricks applications, dashboards, or user-facing data tools.\n\nPlease note: Optiveum acts as the recruitment partner and the direct contracting party for this position. The selected candidate will sign a cooperation agreement directly with Optiveum.","description_format":"text","description_chars":2891,"description_truncated":false,"requirements":{"experience_years_min":5,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[{"language":"English","level":"Advanced (C1)","optional":false}]},"benefits":[],"hiring_locations":[{"name":"Poland","iso":"PL","kind":"country"}],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-10-01T08:06:14Z"}],"visa":[],"liveness":{"score":54,"band":"ok","label":"Likely open","p_open":1,"p_active":0.538,"p_room":1,"age_days":3,"expected_fill_days":19,"reasons":["conf:21","agency","velocity","win:early"],"computed_at":"2026-10-03T05:45:00Z"},"pay":{"stated_usd_annual":86000,"is_top_pay":false},"html_url":"https://alion.io/job/optiveum-senior-data-engineer-databricks-pyspark","json_url":"https://alion.io/job/optiveum-senior-data-engineer-databricks-pyspark.json","meta":{"generated_at":"2026-10-04T02:03:32Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":2799,"day_limit":5000,"remaining_today":2201,"minute_limit":60,"resets_at":"2026-10-05T00:00:00Z"}}}