{"id":1228059,"url":"https://alion.io/job/twinpacs-sdn-bhd-data-engineer","title":"Data Engineer","company":{"id":3800702,"name":"TwinPacs Sdn Bhd","domain":"twinpacs.com","url":"https://alion.io/company/twinpacs-sdn-bhd","size_band":null,"is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":null,"work_mode":"hybrid","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Bengaluru, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":22000,"max_usd":44000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":51},"experience_years_min":6,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Airflow","optional":false},{"name":"Amazon Redshift","optional":false},{"name":"Anomaly Detection","optional":false},{"name":"Ansible","optional":false},{"name":"Apache Iceberg","optional":false},{"name":"Apache Kafka","optional":false},{"name":"AWS","optional":false},{"name":"Azure","optional":false},{"name":"BigQuery","optional":false},{"name":"dbt","optional":false},{"name":"Delta Lake","optional":false},{"name":"Docker","optional":false},{"name":"ETL/ELT","optional":false},{"name":"Flink","optional":false},{"name":"GCP","optional":false},{"name":"Google BigQuery","optional":false},{"name":"Great Expectations","optional":false},{"name":"Java","optional":false},{"name":"Kubernetes","optional":false},{"name":"Python","optional":false},{"name":"Rest API","optional":false},{"name":"Snowflake","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false},{"name":"Terraform","optional":false}],"status":"live","first_seen_at":"2026-09-22T06:39:57Z","employer_posted_date":null,"last_verified_at":"2026-09-22T06:39:57Z","board_verified":false,"closed_at":null,"days_open":9,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":9},"description":"About the Role :\n\nData Engineer to build and maintain scalable, high-performance data pipelines and infrastructure for our next-generation data platform. The platform ingests and processes real-time and historical data from diverse industrial sources such as airport systems, sensors, cameras, and APIs. You will work closely with AI/ML engineers, data scientists, and DevOps to enable reliable analytics, forecasting, and anomaly detection use cases.\n\nKey Responsibilities :\n\n- Design and implement real-time (Kafka, Spark/Flink) and batch (Airflow, Spark) pipelines for high-throughput data ingestion, processing, and transformation.\n\n- Develop data models and manage data lakes and warehouses (Delta Lake, Iceberg, etc) to support both analytical and ML workloads.\n\n- Integrate data from diverse sources: IoT sensors, databases (SQL/NoSQL), REST APIs, and flat files.\n\n- Ensure pipeline scalability, observability, and data quality through monitoring, alerting, validation, and lineage tracking.\n\n- Collaborate with AI/ML teams to provision clean and ML-ready datasets for training and inference.\n\n- Deploy, optimize, and manage pipelines and data infrastructure across on-premise and hybrid environments.\n\n- Participate in architectural decisions to ensure resilient, cost-effective, and secure data flows.\n\n- Contribute to infrastructure-as-code and automation for data deployment using Terraform, Ansible, or similar tools.\n\nQualifications & Required Skills :\n\n- Bachelor's or Master's in Computer Science, Engineering, or related field.\n\n- 6+ years in data engineering roles, with at least 2 years handling real-time or streaming pipelines.\n\n- Strong programming skills in Python/Java and SQL.\n\n- Experience with Apache Kafka, Apache Spark, or Apache Flink for real-time and batch processing.\n\n- Hands-on with Airflow, dbt, or other orchestration tools.\n\n- Familiarity with data modeling (OLAP/OLTP), schema evolution, and format handling (Parquet, Avro, ORC).\n\n- Experience with hybrid/on-prem and cloud platforms (AWS/GCP/Azure) deployments.\n\n- Proficient in working with data lakes/warehouses like Snowflake, BigQuery, Redshift, or Delta Lake.\n\n- Knowledge of DevOps practices, Docker/Kubernetes, Terraform or Ansible.\n\n- Exposure to data observability, data cataloging, and quality tools (e.g., Great Expectations, OpenMetadata).\nSkills\nPython, Apache Spark, SQL, Java, Apache Flink, Delta Lake, ETL, Data Pipeline","description_format":"text","description_chars":2425,"description_truncated":false,"requirements":{"experience_years_min":6,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-09-25T13:06:44Z"}],"liveness":{"score":74,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.781,"p_room":0.945,"age_days":8,"expected_fill_days":24,"reasons":["seen:8","velocity","win:mid"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/twinpacs-sdn-bhd-data-engineer","json_url":"https://alion.io/job/twinpacs-sdn-bhd-data-engineer.json","meta":{"generated_at":"2026-10-01T10:00:29Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":994,"day_limit":5000,"remaining_today":4006,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}