{"id":1593350,"url":"https://alion.io/job/citi-data-engineer-2","title":"Data Engineer","company":{"id":7841,"name":"Citi","domain":"citi.com","url":"https://alion.io/company/citi","size_band":"5000+","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Workday","truth_index":{"grade":"A","score":100,"open_postings":315,"ghost_share":0.01,"stale_share":0.003,"repost_share":0.102,"time_to_fill_p50_days":10,"computed_at":"2026-10-03T05:45:00Z"}},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":"full_time","work_mode":"hybrid","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Pune, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":19000,"max_usd":39000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":51},"experience_years_min":5,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Airflow","optional":false},{"name":"Amazon Redshift","optional":false},{"name":"Amazon S3","optional":false},{"name":"Anomaly Detection","optional":false},{"name":"Apache Kafka","optional":false},{"name":"AWS","optional":false},{"name":"AWS Lambda","optional":false},{"name":"Azure","optional":false},{"name":"BigQuery","optional":false},{"name":"CI/CD","optional":false},{"name":"Dagster","optional":false},{"name":"Databricks","optional":false},{"name":"dbt","optional":false},{"name":"Docker","optional":false},{"name":"DuckDB","optional":false},{"name":"ETL/ELT","optional":false},{"name":"Flink","optional":false},{"name":"GCP","optional":false},{"name":"GDPR","optional":false},{"name":"Google BigQuery","optional":false},{"name":"Great Expectations","optional":false},{"name":"Hadoop","optional":false},{"name":"HIPAA","optional":false},{"name":"Java","optional":false},{"name":"Kubernetes","optional":false},{"name":"Prefect","optional":false},{"name":"pySpark","optional":false},{"name":"Python","optional":false},{"name":"Scala","optional":false},{"name":"SLI/SLO/SLA","optional":false},{"name":"Snowflake","optional":false},{"name":"SOC 2","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false},{"name":"Terraform","optional":false},{"name":"Trino","optional":false}],"status":"closed","first_seen_at":"2026-10-01T16:40:32Z","employer_posted_date":"2026-10-01","last_verified_at":"2026-10-02T06:03:20Z","board_verified":false,"closed_at":"2026-10-02T06:03:20Z","days_open":0,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":0},"description":"We are seeking a skilled and proactive Data Engineer to design, build, and maintain robust, scalable, and optimized data pipelines and architectures. In this role, you will collaborate closely with Data Scientists, Business Intelligence Analysts, and Software Engineers to transform raw enterprise data into clean, accessible, and high-performance data models.\nThe ideal candidate has hands-on experience with modern cloud data platforms, distributed computing, streaming and batch ETL/ELT pipelines, data modeling, and robust data governance practices.\n2. Key Responsibilities\nPipeline Development & Architecture\n Design & Build ETL/ELT Pipelines: Develop scalable batch and real-time streaming data ingestion and transformation pipelines from diverse sources (APIs, transactional DBs, Kafka, logs, flat files).\n Data Modeling & Storage: Architect and optimize dimensional data models (Star/Snowflake schemas, One Big Table), data lakes, data lakehouses, and relational/NoSQL data stores.\n Workflow Orchestration: Implement and maintain pipeline scheduling, dependency management, and monitoring using modern orchestrators (e.g., Apache Airflow, Prefect, Dagster).\nPerformance, Reliability & Scalability\n Query & Storage Optimization: Fine-tune SQL queries, indexing, partitioning, caching, and distributed compute workloads (e.g., Spark, Trino, DuckDB).\n Data Quality & Testing: Implement automated data validation, anomaly detection, and schema evolution testing (e.g., Great Expectations, Soda, dbt tests).\nMonitoring & Alerting: Build observability frameworks for pipeline uptime, latency, SLA adherence, and data freshness.\nCollaboration & Governance\n Data Governance & Security: Enforce enterprise data security, role-based access control (RBAC), data lineage, and compliance standards (GDPR, CCPA, SOC2, HIPAA).\n Stakeholder Enablement: Partner with downstream data consumers (BI developers, ML engineers, business stakeholders) to understand data requirements and deliver self-service datasets.\n CI/CD & DevOps: Champion Infrastructure as Code (Terraform), containerization (Docker, Kubernetes), and CI/CD best practices for data products.\n3. Required Qualifications & Skills\nTechnical Competencies\nProgramming Languages: Proficiency in Python and/or Scala/Java, with strong knowledge of software engineering best practices (OOP, clean code, unit testing, git).\n SQL & Data Warehousing: Advanced SQL proficiency; deep expertise in modern cloud data warehouses (e.g., Snowflake, Google BigQuery, AWS Redshift, Databricks Lakehouse).\nDistributed Computing: Hands-on experience with Apache Spark / PySpark, Apache Flink, or Hadoop ecosystems.\nData Transformation & Orchestration: Experience with modern data stack tools such as dbt (data build tool) and Apache Airflow.\n Cloud Infrastructure: Solid experience with at least one major cloud provider (AWS, GCP, or Azure) and core data services (e.g., S3/GCS/ADLS, Glue, EMR, BigQuery, Lambda/Cloud Functions).\n5+ years of experience leading complex data platform implementations and architecture decisions.\n------------------------------------------------------\nJob Family Group:\nTechnology------------------------------------------------------\nJob Family:\nApplications Development------------------------------------------------------\nTime Type:\nFull time------------------------------------------------------\nMost Relevant Skills\nPlease see the requirements listed above.------------------------------------------------------\nOther Relevant Skills\nFor complementary skills, please see above and/or contact the recruiter.------------------------------------------------------\nCiti is an equal opportunity employer, and qualified candidates will receive consideration without regard to their race, color, religion, sex, sexual orientation, gender identity, national origin, disability, status as a protected veteran, or any other characteristic protected by law.\nIf you are a person with a disability and need a reasonable accommodation to use our search tools and/or apply for a career opportunity review Accessibility at Citi.\nView Citi’s EEO Policy Statement and the Know Your Rights poster.","description_format":"text","description_chars":4149,"description_truncated":false,"requirements":{"experience_years_min":5,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[{"name":"India","iso":"IN","kind":"country"}],"hiring_excludes":[],"relocation_offered":false,"industries":["Financial Services","Wealth Management & Financial Advisors","Payment Processing & Gateways","Consumer Loans & Pawnshops"],"lifecycle":[{"event":"open","at":"2026-10-01T16:40:32Z"},{"event":"close","at":"2026-10-02T06:03:20Z"}],"visa":[],"liveness":null,"pay":null,"html_url":"https://alion.io/job/citi-data-engineer-2","json_url":"https://alion.io/job/citi-data-engineer-2.json","meta":{"generated_at":"2026-10-04T00:21:54Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":321,"day_limit":5000,"remaining_today":4679,"minute_limit":60,"resets_at":"2026-10-05T00:00:00Z"}}}