{"id":1366506,"url":"https://alion.io/job/gea-data-engineer","title":"Data Engineer","company":{"id":1788131,"name":"GEA","domain":"gea.com","url":"https://alion.io/company/gea-com","size_band":"5000+","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Workday","truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bogotá, Colombia","Campinas, Brazil"],"countries":["BR"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":56000,"max_usd":146000,"period":"year","method":"global_role_cell_scaled_by_country","sample_n":1536},"experience_years_min":5,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Agile","optional":false},{"name":"Anomaly Detection","optional":false},{"name":"Azure Data Factory","optional":false},{"name":"Data Vault","optional":false},{"name":"Databricks","optional":false},{"name":"Dimensional Modeling","optional":false},{"name":"ETL/ELT","optional":false},{"name":"SQL","optional":false}],"status":"live","first_seen_at":"2026-09-21T00:00:00Z","employer_posted_date":"2026-09-21","last_verified_at":"2026-09-29T23:25:53Z","board_verified":true,"closed_at":null,"days_open":9,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":9},"description":"GEA is one of the world’s largest systems suppliers for the food, beverage and pharmaceutical sectors. Our portfolio includes machinery and plants as well as advanced process technology, components and comprehensive services. Used across diverse industries, they enhance the sustainability and efficiency of production processes globally.We are looking for a Data Engineer\nResponsibilities / Tasks\nData Architecture & Infrastructure\nDesign and implement a unified data warehouse and/or data lake capable of serving multiple analytics and AI workloads.\nDefine the overall data architecture strategy, including storage layers, access patterns, and scalability approach.\nDefine the data extraction and landing strategy, partitioning and SCD (slowly changing dimensions), data modelling strategy and consumption ports.\nPipeline Development & Integration\nBuild and maintain ETL/ELT pipelines consuming data from multiple enterprise source systems (ERP, CRM, operational tools, and others).\nDevelop pipelines using Databricks (Lakeflow Connect) & Azure Data Factory as primary platforms.\nEnsure pipeline reliability, scalability, and observability through monitoring, alerting, and logging.\nData Quality & Governance\nEstablish and enforce data quality standards, validation rules, and anomaly detection processes.\nImplement data cataloguing, lineage tracking, and documentation practices to ensure transparency and auditability.\nDefine naming conventions, schema standards, and access control policies in coordination with stakeholders.\nCollaboration & Stakeholder Engagement\nWork closely with data scientists, BI analysts, and developers within the team to ensure data products meet downstream requirements.\nTranslate business requirements from non-technical stakeholders into robust data models and pipeline logic.\nActively contribute to sprint planning and technical decision-making within an agile team environment.\nContinuous Improvement\nMonitor and optimize query performance, pipeline efficiency, and infrastructure cost.\nStay current with developments in data engineering tooling, cloud platforms, and best practices.\nContribute to the team's knowledge base through documentation and internal knowledge-sharing.\nYour Profile / Qualifications\nMinimum 5 years of professional experience in data engineering or a closely related field.\nExpert-level proficiency in SQL - including complex query design, performance tuning, and schema modeling.\nHands-on experience with Databricks for large-scale data processing and pipeline orchestration.\nProven experience designing and implementing data warehouse or data lake solutions at enterprise scale.\nStrong understanding of ETL/ELT design patterns, data modeling methodologies (star schema, data vault, etc.), and pipeline orchestration.\nExperience integrating data from heterogeneous source systems (ERP platforms, APIs, flat files, operational databases).\nAbility to communicate technical concepts clearly to both technical and non-technical audiences.\nProfessional-level proficiency in Spanish; working English is a strong advantage.\nDid we spark your interest?\nThen please click apply above to access our guided application process.","description_format":"text","description_chars":3178,"description_truncated":false,"requirements":{"experience_years_min":5,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[{"language":"Spanish","level":"Advanced (C1)","optional":false}]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Food & Beverages","Health Care","Food Technology"],"lifecycle":[{"event":"open","at":"2026-09-28T02:24:17Z"}],"liveness":{"score":75,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.835,"p_room":0.9,"age_days":8,"expected_fill_days":22,"reasons":["conf:1","velocity","win:mid","comp:brand"],"computed_at":"2026-09-29T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/gea-data-engineer","json_url":"https://alion.io/job/gea-data-engineer.json","meta":{"generated_at":"2026-09-30T02:07:22Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1526,"day_limit":5000,"remaining_today":3474,"minute_limit":60,"resets_at":"2026-10-01T00:00:00Z"}}}