{"id":1292882,"url":"https://alion.io/job/nasugroup-senior-data-engineer","title":"Senior Data Engineer","company":{"id":3800363,"name":"Nasugroup","domain":"nasugroup.com","url":"https://alion.io/company/nasugroup","size_band":null,"is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bengaluru, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":22000,"max_usd":45000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":54},"experience_years_min":5,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AI Agents","optional":false},{"name":"Airflow","optional":false},{"name":"Amazon Redshift","optional":false},{"name":"Amazon S3","optional":false},{"name":"AWS","optional":false},{"name":"Azure","optional":false},{"name":"BigQuery","optional":false},{"name":"CI/CD","optional":false},{"name":"Dagster","optional":false},{"name":"Dask","optional":false},{"name":"Databricks","optional":false},{"name":"Dimensional Modeling","optional":false},{"name":"Docker","optional":false},{"name":"Embeddings","optional":false},{"name":"Feature Store","optional":false},{"name":"FSDP","optional":false},{"name":"GCP","optional":false},{"name":"Git","optional":false},{"name":"Google BigQuery","optional":false},{"name":"Machine Learning","optional":false},{"name":"ONNX Runtime","optional":false},{"name":"Prefect","optional":false},{"name":"Python","optional":false},{"name":"PyTorch","optional":false},{"name":"RAG","optional":false},{"name":"Ray","optional":false},{"name":"Snowflake","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false},{"name":"TorchServe","optional":false},{"name":"Triton","optional":false},{"name":"Amazon Kinesis","optional":true},{"name":"Amazon SageMaker","optional":true},{"name":"Apache Hudi","optional":true},{"name":"Apache Iceberg","optional":true},{"name":"Apache Kafka","optional":true},{"name":"dbt","optional":true},{"name":"Delta Lake","optional":true},{"name":"Fine-tuning","optional":true},{"name":"Flink","optional":true},{"name":"Hugging Face","optional":true},{"name":"Kubeflow","optional":true},{"name":"Kubernetes","optional":true},{"name":"LLM","optional":true},{"name":"LLM Evaluation","optional":true},{"name":"LoRA","optional":true},{"name":"MLFlow","optional":true},{"name":"PEFT","optional":true},{"name":"pgvector","optional":true},{"name":"Pinecone","optional":true},{"name":"PostgreSQL","optional":true},{"name":"Qdrant","optional":true},{"name":"QLoRA","optional":true},{"name":"Terraform","optional":true},{"name":"Transformers","optional":true},{"name":"Vertex AI","optional":true},{"name":"Weaviate","optional":true},{"name":"Weights & Biases","optional":true}],"status":"live","first_seen_at":"2026-08-11T05:41:59Z","employer_posted_date":null,"last_verified_at":"2026-08-11T05:41:59Z","board_verified":false,"closed_at":null,"days_open":56,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":56},"description":"About the Role :\n\nWe are looking for a Senior Data Engineer who is equally comfortable with production data pipelines and with PyTorch. You will own the data layer that our machine learning and agentic AI systems depend on ingestion, transformation, feature engineering, training data curation, and the infrastructure that gets models from a notebook into production.\n\nThis is a hands-on engineering role, not a supervisory one. You will write code every day, work directly with client engineering teams, and be accountable for the reliability and cost of what you build.\n\nRoles and Responsibilities :\n\n- Design, build and operate batch and streaming data pipelines that feed model training and inference at production scale.\n\n- Build and optimise PyTorch training and inference workflows custom Datasets and DataLoaders, distributed training (DDP/FSDP), mixed precision, checkpointing and reproducibility.\n\n- Own feature engineering and feature store design; ensure training/serving parity and prevent data leakage.\n\n- Curate, version and validate training datasets including labelling workflows, data quality checks and drift detection.\n\n- Deploy and serve models (TorchServe, ONNX Runtime, Triton or equivalent) with sensible latency, throughput and cost characteristics.\n\n- Build data infrastructure for retrieval-augmented and agentic systems: embedding pipelines, vector stores, chunking strategies and evaluation datasets.\n\n- Instrument everything pipeline observability, lineage, data contracts, model performance monitoring and alerting.\n\n- Partner with frontend, backend and platform engineers to ship end-to-end AI features, and with client stakeholders to translate ambiguous requirements into a data design.\n\n- Review code, raise the engineering bar, and mentor mid-level engineers on the team.\n\nMust-Have Qualifications :\n\n- 5+ years of professional experience in data engineering, ML engineering or a closely related discipline.\n\n- Strong, production-grade Python. You write tested, typed, maintainable code not just scripts.\n\n- Hands-on PyTorch experience: building and training models, writing custom data loading, debugging training runs, and taking at least one model to production.\n\n- Deep SQL and solid data modelling fundamentals (dimensional modelling, partitioning, indexing, query optimisation).\n\n- Production experience with a distributed processing engine Apache Spark, Ray, Dask or equivalent.\n\n- Orchestration experience with Airflow, Dagster, Prefect or similar, including backfills, idempotency and failure recovery.\n\n- Cloud data platform experience on AWS, GCP or Azure (S3/GCS, Glue/Dataproc, EMR, Redshift/BigQuery/Snowflake, Databricks or comparable).\n\n- Comfortable with Docker, Git-based workflows and CI/CD; able to containerise and ship your own work.\n\n- Clear written and spoken English you will be writing design docs and talking to enterprise clients.\n\nGood to Have :\n\n- Experience with LLM and agentic systems embeddings, vector databases (pgvector, Pinecone, Weaviate, Qdrant), RAG pipelines, or evaluation harnesses for LLM outputs.\n\n- Fine-tuning or parameter-efficient tuning experience (LoRA, QLoRA) and familiarity with Hugging Face Transformers.\n\n- MLOps tooling : MLflow, Weights & Biases, Kubeflow, SageMaker or Vertex AI.\n\n- Kubernetes, Terraform or other infrastructure-as-code experience.\n\n- Streaming systems: Kafka, Kinesis, Flink or Spark Structured Streaming.\n\n- Lakehouse formats Delta Lake, Apache Iceberg or Hudi.\n\n- dbt and modern analytics engineering practice.\n\n- GPU performance tuning, quantisation, or inference cost optimisation.\n\n- Open-source contributions or published technical writing.\nSkills\nData Engineering, Data Pipeline, PyTorch, Machine Learning, SQL, Data Modeling, Apache Airflow, Apache Spark, AWS, LLM, Data Build Tool","description_format":"text","description_chars":3803,"description_truncated":false,"requirements":{"experience_years_min":5,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-09-26T08:01:01Z"}],"visa":[],"liveness":{"score":7,"band":"cold","label":"Long shot","p_open":0.4,"p_active":0.535,"p_room":0.35,"age_days":56,"expected_fill_days":23,"reasons":["seen:56","velocity","win:tail"],"computed_at":"2026-10-06T05:45:30Z"},"pay":null,"html_url":"https://alion.io/job/nasugroup-senior-data-engineer","json_url":"https://alion.io/job/nasugroup-senior-data-engineer.json","meta":{"generated_at":"2026-10-06T23:21:09Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","about":"Alion is a live layer of people, companies and AI agents: who they are, whether they are real and active right now, what they do and how to work with them, readable by people and by agents and paid per call.","catalog":"https://alion.io/catalog.json","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4648,"day_limit":5000,"remaining_today":352,"minute_limit":60,"resets_at":"2026-10-07T00:00:00Z"}},"offers":[{"id":"company.slices","title":"One company in depth, by slice","status":"live","price":{"credits":0.02,"usd":0.002,"plus_per_slice":{"credits":0.05,"usd":0.005}},"unit":"per company, plus each slice with data","note":"the employer in depth","call":{"mcp_tool":"get_company","arguments":{"id":3800363},"rest":"https://alion.io/mcp/rest/get_company?id=3800363"},"human":"https://alion.io/catalog?offer=company.slices&for=job%2Fnasugroup-senior-data-engineer"},{"id":"market.stats","title":"A market slice: pay, demand and time to fill","status":"live","price":{"credits":1,"usd":0.1},"unit":"per slice","note":"pay, demand and time to fill for this role and place","call":{"mcp_tool":"market_stats"},"human":"https://alion.io/catalog?offer=market.stats&for=job%2Fnasugroup-senior-data-engineer"},{"id":"job.search","title":"Open jobs by role, technology, place, pay and visa","status":"live","price":{"credits":0.02,"usd":0.002},"unit":"per posting in a list","note":"similar open postings","call":{"mcp_tool":"search_jobs"},"human":"https://alion.io/catalog?offer=job.search&for=job%2Fnasugroup-senior-data-engineer"},{"id":"company.verify","title":"Is this company real and active right now","status":"pilot","price":null,"unit":"per company","request":{"url":"https://alion.io/catalog/request","method":"POST","body":"{\"offer\": \"company.verify\", \"for\": \"job/nasugroup-senior-data-engineer\", \"note\": \"what you need it for\"}"},"human":"https://alion.io/catalog?offer=company.verify&for=job%2Fnasugroup-senior-data-engineer"}]}