{"id":1458817,"url":"https://alion.io/job/infosys-ai-data-engineer","title":"AI Data Engineer","company":{"id":223,"name":"Infosys","domain":"infosys.com","url":"https://alion.io/company/infosys","size_band":"5000+","is_staffing_agency":false,"employer_type":"services","is_intermediary":false,"listed_via":null,"ats_vendor":"Career site","truth_index":{"grade":"B","score":75,"open_postings":419,"ghost_share":0,"stale_share":1,"repost_share":0,"time_to_fill_p50_days":null,"computed_at":"2026-10-01T05:45:00Z"}},"role":"Data Science","role_family":"Data Science","seniority":"middle","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bengaluru, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":15500,"max_usd":38000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":13},"experience_years_min":3,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"ETL/ELT","optional":false},{"name":"Agile","optional":true},{"name":"Airflow","optional":true},{"name":"Databricks","optional":true},{"name":"Explainable AI","optional":true},{"name":"Python","optional":true},{"name":"Spark","optional":true},{"name":"SQL","optional":true}],"status":"live","first_seen_at":"2026-09-19T08:36:58Z","employer_posted_date":"2026-09-29","last_verified_at":"2026-09-29T17:24:37Z","board_verified":true,"closed_at":null,"days_open":12,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":12},"description":"About the job:\nJoin a team where data engineering meets intelligent automation to power next-generation analytics and AI experiences. In this role, you’ll help build reliable, scalable data pipelines and enable GenAI-ready datasets that support experimentation, model development, and business decision-making. You’ll collaborate closely with analysts, data scientists, and platform teams to turn raw, fragmented data into trusted, well-modeled, and accessible assets. If you enjoy solving complex data challenges, improving performance, and bringing structure to fast-moving AI initiatives, this is a great opportunity to grow your impact. You’ll work in a culture that values ownership, continuous learning, and practical innovation-where clean data, strong engineering, and thoughtful collaboration come together to deliver real outcomes.\nResponsibilities\nKey Responsibilities:\nDesign, build, and maintain robust ETL/ELT pipelines to ingest, transform, and curate data from multiple sources.\nDevelop and optimize data models and curated datasets to support analytics, reporting, and AI/ML workloads.\nImplement data quality checks, validation rules, and monitoring to ensure accuracy, completeness, and reliability.\nEnable GenAI initiatives by preparing high-quality datasets for downstream consumption (e.g., feature-ready and retrieval-ready data).\nCollaborate with cross-functional teams to gather requirements, define data contracts, and deliver reusable data assets.\nTroubleshoot pipeline failures and performance bottlenecks; improve scalability, latency, and cost efficiency.\nMaintain documentation for pipelines, transformations, lineage, and operational runbooks to support maintainability.\nMinimum Qualifications:\nBTECH, MTECH, MCA, MSC or equivalent education.\n3-5 years of experience in data engineering with hands-on ownership of production-grade pipelines.\nStrong experience in ETL processes including extraction, transformation, orchestration, and scheduling.\nWorking exposure to GenAI-oriented data preparation needs and supporting AI/ML data workflows.\nAbility to collaborate with stakeholders to translate requirements into scalable data solutions.\nTechnical requirements\nGood to have skills:\nSQL, Python, Apache Spark, Airflow, Data Modeling, data engineering, ai\nAdditional responsibilities\nPreferred Qualifications:\nExperience designing scalable data architectures and implementing reusable data frameworks for multiple use cases.\nFamiliarity with building datasets for GenAI use cases such as retrieval workflows and knowledge augmentation patterns.\nProven ability to improve pipeline reliability through automation, alerting, and proactive monitoring.\nStrong problem-solving skills with a track record of optimizing transformations and reducing end-to-end processing time.\nExperience working in agile teams and contributing to code reviews, documentation, and engineering best practices.\nPreferred skills\nTechnology->Data Engineering->Databricks,Technology->AI-Responsible AI->Responsible AI->explainable ai\nEducation\nMCA,MSc,MTech,Bachelor of Engineering,BTech","description_format":"text","description_chars":3086,"description_truncated":false,"requirements":{"experience_years_min":3,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":false},"security_clearance":false,"languages":[]},"benefits":["Continuous learning"],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["IT Consulting & Digital Transformation","IT Outsourcing & Dedicated Teams"],"lifecycle":[{"event":"open","at":"2026-09-29T11:08:16Z"}],"liveness":{"score":53,"band":"ok","label":"Likely open","p_open":1,"p_active":0.583,"p_room":0.9,"age_days":11,"expected_fill_days":18,"reasons":["conf:36","stale_co","velocity","win:mid","comp:brand"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/infosys-ai-data-engineer","json_url":"https://alion.io/job/infosys-ai-data-engineer.json","meta":{"generated_at":"2026-10-01T10:44:26Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1923,"day_limit":5000,"remaining_today":3077,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}