{"id":1611602,"url":"https://alion.io/job/cummins-ai-data-engineer-senior","title":"AI Data Engineer - Senior","company":{"id":1758413,"name":"Cummins","domain":"cummins.com","url":"https://alion.io/company/cummins-com","size_band":"1001-5000","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Oracle","truth_index":{"grade":"A","score":91,"open_postings":24,"ghost_share":0,"stale_share":0.542,"repost_share":0,"time_to_fill_p50_days":7,"computed_at":"2026-10-07T05:47:15Z"}},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Pune, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":19000,"max_usd":38000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":54},"experience_years_min":5,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Agile","optional":false},{"name":"Alation","optional":false},{"name":"Apache Kafka","optional":false},{"name":"Azure","optional":false},{"name":"Cassandra","optional":false},{"name":"CI/CD","optional":false},{"name":"DynamoDB","optional":false},{"name":"ETL/ELT","optional":false},{"name":"Hadoop","optional":false},{"name":"HBase","optional":false},{"name":"Java","optional":false},{"name":"Kanban","optional":false},{"name":"Machine Learning","optional":false},{"name":"Neo4j","optional":false},{"name":"pySpark","optional":false},{"name":"Python","optional":false},{"name":"Scala","optional":false},{"name":"Scrum","optional":false},{"name":"Snowflake","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false},{"name":"Time Series Forecasting","optional":false}],"status":"live","first_seen_at":"2026-10-01T13:53:46Z","employer_posted_date":"2026-10-01","last_verified_at":"2026-10-08T00:03:00Z","board_verified":true,"closed_at":null,"days_open":6,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":6},"description":"Job Summary:\nLeads projects for design, development and maintenance of a data and analytics platform. Effectively and efficiently process, store and make data available to analysts and other consumers. Works with key business stakeholders, IT experts and subject-matter experts to plan, design and deliver optimal analytics and data science solutions. Works on one or many product teams at a time.\nKey Responsibilities:\nDesigns and automates deployment of our distributed system for ingesting and transforming data from various types of sources (relational, event-based, unstructured). Designs and implements framework to continuously monitor and troubleshoot data quality and data integrity issues. Implements data governance processes and methods for managing metadata, access, retention to data for internal and external users. Designs and provide guidance on building reliable, efficient, scalable and quality data pipelines with monitoring and alert mechanisms that combine a variety of sources using ETL/ELT tools or scripting languages. Designs and implements physical data models to define the database structure. Optimizing database performance through efficient indexing and table relationships. Participates in optimizing, testing, and troubleshooting of data pipelines. Designs, develops and operates large scale data storage and processing solutions using different distributed and cloud based platforms for storing data (e.g. Data Lakes, Hadoop, Hbase, Cassandra, MongoDB, Accumulo, DynamoDB, others). Uses innovative and modern tools, techniques and architectures to partially or completely automate the most-common, repeatable and tedious data preparation and integration tasks in order to minimize manual and error-prone processes and improve productivity. Assists with renovating the data management infrastructure to drive automation in data integration and management. Ensures the timeliness and success of critical analytics initiatives by using agile development technologies such as DevOps, Scrum, Kanban Coaches and develops less experienced team members.\nCompetencies: Security & Compliance Principles - Applies standards, tools, and best practices to embed security, privacy, and compliance into the design/build/test/operate lifecycle for products, services, apps, systems, software, and configurations-balancing protection, efficiency, and cost.\nProgramming Principles - Applies programming languages, frameworks, and patterns to design, write, configure, test, and maintain software/solutions/systems that are efficient, secure, scalable, and reliable.\nData Principles - Governs, models, secures, implements, and observes data flows to ensure integrity, quality, and compliance-enabling trusted, scalable, cost-conscious data use.\nModern Development Practices - Applies modern engineering practices and tools-such as Agile/DevSecOps, CI/CD, automated testing, and infrastructure as code-to accelerate delivery, improve quality, and reduce risk across the SDLC.\nSolution Design - Translate business requirements into integrated designs, architectures, patterns, and system interactions that deliver customer value and align with enterprise standards and subject-matter platforms.\nDemonstrating Mastery - Maintains essential knowledge and proficiency in relevant domains, tools, technologies, methodologies, or frameworks through targeted credentials and rigorous proficiency, future-proofing organizational skills against strategic needs.\nStrategic and Innovative Thinking - Evaluates business and technology trends, anticipates future needs, develops creative approaches, and frames innovations to shape strategy and create durable value with cost-aware innovation.\nTechnical Passion & Drive - Models curiosity and excitement for technology by self-initiating continuous development, experimenting with emerging technologies, and identifying insertion opportunities that accelerate business performance.\nDriving Effective Outcomes - Takes ownership, acts with urgency, and initiates action to turn goals into clear plans, decisions, guardrails, and cadences while navigating ambiguity and change to drive momentum and deliver consistent results.\nEngaging with Impact - Communicates with clarity and purpose to align stakeholders, foster collaboration, build trust, and influence coordinated action across teams and functions to accelerate outcomes.\nValues Differences - Recognizing the value that different perspectives and cultures bring to an organization.\nEnsuring Customer Success - Embraces a customer-first mindset to deliver outcomes by linking customer needs and business priorities to aligned solutions, delivery, adoption, satisfaction, and realized value through sustained engagement that builds partnership and trust.\nEducation, Licenses, Certifications: College, university, or equivalent degree in relevant technical discipline, or relevant equivalent experience required. This position may require licensing for compliance with export controls or sanctions regulations.\nExperience: Intermediate experience in a relevant discipline area is required. Knowledge of the latest technologies and trends in data engineering are highly preferred and includes:\n- Familiarity analyzing complex business systems, industry requirements, and/or data regulations\n- Background in processing and managing large data sets\n- Design and development for a Big Data platform using open source and third-party tools\n- SPARK, Scala/Java, Map-Reduce, Hive, Hbase, and Kafka or equivalent college coursework\n- SQL query language\n- Clustered compute cloud-based implementation experience\n- Experience developing applications requiring large file movement for a Cloud-based environment and other data extraction tools and methods from a variety of sources\n- Experience in building analytical solutions\nIntermediate experiences in the following are preferred:\n- Experience with IoT technology\n- Experience in Agile software development\n- Experience with continuous improvement across cost optimization, performance tuning and scalability of Data Engineering pipelines.\n- Experience with enabling self service data engineering pipeline implementation capabilities for end users preferred.\n- Experience with using Co-pilot/AI capabilities to improve the productivity of Data Engineering pipeline development/testing activities.\n Experience:\n5 to 8 years of experience in data engineering, with strong expertise in building and optimizing scalable data pipelines, ETL/ELT processes, and data integration solutions. Skilled in designing robust architectures that support advanced analytics, reporting, and data-driven applications.\nTechnical Skills:\nRequired:\nKnowledge of the latest technologies and trends in data science is highly preferred.\n\nHands on experiences in the following are preferred:\n- Exposure to Big Data open source\n- Clustered compute cloud-based implementation experience\n\nFamiliarity analyzing complex business systems, industry requirements, and/or data regulations\n\nUnderstanding of AI/ML concepts and tools\n\nExperience in ETL/ELT Data Engineering Technologies\n\nBackground in processing and managing large data sets\n\nDesign and development for a Big Data platform using open source and third-party tools\n\nProficiency in Python, SQL, and Spark (PySpark preferred).\n\nHands-on experience integrating with platforms like Palantir, Snowflake, Neo4j, etc.\n\nSolid knowledge of machine learning workflows, model deployment, and advanced analytics (regression, clustering, time-series analysis).\n\nUnderstanding of data governance, data cataloging tools (e.g., Azure Purview, Alation), and metadata management.\n\nSQL query language\n\nClustered compute cloud-based implementation experience\n\nExperience developing applications requiring large file movement for a Cloud-based environment and other data extraction tools and methods from a variety of sources\n\nTake full ownership of the developed data pipelines, providing ongoing support for enhancements and performance optimization\n\nNice to have:\nExperience with graph data modeling and graph databases (e.g., Neo4j, TigerGraph) and familiarity with Palantir Ontology is a strong plus\n\nUnderstanding of data governance, data cataloging tools (e.g., Azure Purview, Alation), and metadata management.\n\nAdditional Key Responsibilities:\nStay current with AI trends and suggest improvements to existing systems and workflows\nExcellent verbal and written communication skills\nDemonstrated self-starter with a proactive, problem-solving mindset\nCandidate need to work from Cummins Pune IOC-B office for 3 days a week. There will be an overlap of few hours with US EST time zone.","description_format":"text","description_chars":8644,"description_truncated":false,"requirements":{"experience_years_min":5,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-10-01T20:20:23Z"}],"visa":[],"liveness":{"score":43,"band":"fade","label":"Fading","p_open":1,"p_active":0.569,"p_room":0.75,"age_days":5,"expected_fill_days":7,"reasons":["conf:1","stale_co","velocity","win:late","comp:brand"],"computed_at":"2026-10-07T05:47:15Z"},"pay":null,"html_url":"https://alion.io/job/cummins-ai-data-engineer-senior","json_url":"https://alion.io/job/cummins-ai-data-engineer-senior.json","meta":{"generated_at":"2026-10-08T00:19:39Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","about":"Alion is a live layer of people, companies and AI agents: who they are, whether they are real and active right now, what they do and how to work with them, readable by people and by agents and paid per call.","catalog":"https://alion.io/catalog.json","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":281,"day_limit":5000,"remaining_today":4719,"minute_limit":60,"resets_at":"2026-10-09T00:00:00Z"}},"offers":[{"id":"company.slices","title":"One company in depth, by slice","status":"live","price":{"credits":0.02,"usd":0.002,"plus_per_slice":{"credits":0.05,"usd":0.005}},"unit":"per company, plus each slice with data","note":"the employer in depth","call":{"mcp_tool":"get_company","arguments":{"id":1758413},"rest":"https://alion.io/mcp/rest/get_company?id=1758413"},"human":"https://alion.io/catalog?offer=company.slices&for=job%2Fcummins-ai-data-engineer-senior"},{"id":"market.stats","title":"A market slice: pay, demand and time to fill","status":"live","price":{"credits":1,"usd":0.1},"unit":"per slice","note":"pay, demand and time to fill for this role and place","call":{"mcp_tool":"market_stats"},"human":"https://alion.io/catalog?offer=market.stats&for=job%2Fcummins-ai-data-engineer-senior"},{"id":"job.search","title":"Open jobs by role, technology, place, pay and visa","status":"live","price":{"credits":0.02,"usd":0.002},"unit":"per posting in a list","note":"similar open postings","call":{"mcp_tool":"search_jobs"},"human":"https://alion.io/catalog?offer=job.search&for=job%2Fcummins-ai-data-engineer-senior"},{"id":"company.verify","title":"Is this company real and active right now","status":"pilot","price":null,"unit":"per company","request":{"url":"https://alion.io/catalog/request","method":"POST","body":"{\"offer\": \"company.verify\", \"for\": \"job/cummins-ai-data-engineer-senior\", \"note\": \"what you need it for\"}"},"human":"https://alion.io/catalog?offer=company.verify&for=job%2Fcummins-ai-data-engineer-senior"}]}