{"id":1558817,"url":"https://alion.io/job/convegenius-junior-data-engineer-2","title":"Junior Data Engineer","company":{"id":1281,"name":"ConveGenius","domain":"convegenius.com","url":"https://alion.io/company/convegenius","size_band":"201-500","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Keka","truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"junior","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Shimla, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":{"min":1200000,"max":1500000,"currency":"INR","period":"year","gross":null,"usd_annual":15710},"salary_estimate":null,"experience_years_min":3,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Amazon Redshift","optional":false},{"name":"Amazon S3","optional":false},{"name":"AWS","optional":false},{"name":"AWS Lambda","optional":false},{"name":"AWS Step Functions","optional":false},{"name":"CI/CD","optional":false},{"name":"Delta Lake","optional":false},{"name":"ETL/ELT","optional":false},{"name":"Git","optional":false},{"name":"pySpark","optional":false},{"name":"Python","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false},{"name":"Azure","optional":true},{"name":"Azure Data Factory","optional":true},{"name":"BigQuery","optional":true},{"name":"Databricks","optional":true},{"name":"GCP","optional":true},{"name":"Google BigQuery","optional":true},{"name":"MySQL","optional":true},{"name":"PostgreSQL","optional":true}],"status":"live","first_seen_at":"2026-09-30T08:11:15Z","employer_posted_date":"2026-09-30","last_verified_at":"2026-10-01T13:46:48Z","board_verified":true,"closed_at":null,"days_open":1,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":1},"description":"Role Overview\nWe are looking for a Junior Data Engineer to join our Data Platform team. You will design and maintain scalable data pipelines and architectures using AWS services, enabling reliable data movement, transformation, and analytics at scale. You will collaborate with analytics, product, and engineering teams to support reporting, dashboards, and insights for millions of students and schools.\nKey Responsibilities\nDesign, build, and maintain ETL/ELT pipelines for large-scale data ingestion, transformation, and loading.\nDevelop and optimize Spark and PySpark jobs for batch and real-time data processing.\nWork with AWS services (S3, Glue, Lambda, Redshift, Athena, EMR) to manage the data ecosystem.\nSupport the design and implementation of Data Lake and Data Warehouse architectures.\nImplement data validation, partitioning, and schema management for efficient query performance.\nCollaborate with data analysts and BI teams to ensure data availability and consistency.\nMaintain data lineage, metadata, data quality, and governance.\nImplement monitoring and alerting for data pipelines.\nUse Git and CI/CD tools to manage code and automate deployment of data workflows.\nQualifications\nEducation: Bachelor's degree in Computer Science, Information Technology, Data Engineering, or a related field.\nExperience: 3-5 years of hands-on experience in data engineering, pipeline development, or cloud-based data systems.\nCore Skills: Strong knowledge of SQL and Python/PySpark.\nCloud & Data Stack: Practical experience with AWS (S3, Glue, Lambda, Redshift, Athena, EMR, Step Functions).\nArchitecture: Solid understanding of Data Lake architecture, ETL/ELT frameworks, and data warehousing concepts.\nFrameworks: Familiarity with Delta Lake, Spark SQL, or other big data frameworks, along with data modeling and performance tuning.\nNice to Have\nExposure to GCP (BigQuery, Dataflow) or Azure (Data Factory, Synapse, Databricks).\nExperience with Databricks, PostgreSQL, MySQL, NoSQL, or Airflow.\nUnderstanding of DevOps practices, CI/CD pipelines, and infrastructure automation.\nPrior experience in an EdTech or public data ecosystem.","description_format":"text","description_chars":2135,"description_truncated":false,"requirements":{"experience_years_min":3,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":false},"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Primary Education"],"lifecycle":[{"event":"open","at":"2026-10-01T03:17:37Z"}],"liveness":{"score":90,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.903,"p_room":1,"age_days":0,"expected_fill_days":63,"reasons":["conf:2","velocity","win:early","comp:junior"],"computed_at":"2026-10-01T05:45:00Z"},"pay":{"stated_usd_annual":15710,"is_top_pay":false},"html_url":"https://alion.io/job/convegenius-junior-data-engineer-2","json_url":"https://alion.io/job/convegenius-junior-data-engineer-2.json","meta":{"generated_at":"2026-10-01T21:16:15Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"assistant","counted_by":"address","units_charged":1,"used_today":256,"day_limit":2000,"remaining_today":1744,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}