{"id":1230763,"url":"https://alion.io/job/codvo-data-engineer","title":"Data Engineer","company":{"id":3800246,"name":"Codvo","domain":"codvo.ai","url":"https://alion.io/company/codvo","size_band":null,"is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":null,"truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Pune, India"],"countries":["IN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":20000,"max_usd":41000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":51},"experience_years_min":7,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Amazon S3","optional":false},{"name":"AWS","optional":false},{"name":"AWS Lambda","optional":false},{"name":"CI/CD","optional":false},{"name":"Databricks","optional":false},{"name":"Delta Lake","optional":false},{"name":"ETL/ELT","optional":false},{"name":"IAM","optional":false},{"name":"Machine Learning","optional":false},{"name":"Python","optional":false},{"name":"Spark","optional":false},{"name":"FastAPI","optional":true},{"name":"Flask","optional":true},{"name":"pySpark","optional":true}],"status":"live","first_seen_at":"2026-09-17T10:56:23Z","employer_posted_date":null,"last_verified_at":"2026-09-17T10:56:23Z","board_verified":false,"closed_at":null,"days_open":13,"trust":{"level":"not_scored","repost_count":null,"flags":[],"days_open":13},"description":"Job Description : \n\n- Design, build, and maintain Databricks data pipelines (ETL/ELT) for ingestion, transformation, and orchestration using Spark/Delta Lake/Databricks Workflows.\n\n- Operationalize machine learning models by building inference pipelines that invoke models authored by data scientists (batch or real-time), ensuring consistency between training and inference environments.\n\n- Ensure data reliability, quality, and observability through robust validation, monitoring, alerting, and automated recovery mechanisms.\n\n- Collaborate closely with data scientists to productionize models, manage model deployment lifecycles, and optimize inference performance and cost.\n\n- Implement best-practice DevOps/MLOps processes such as CI/CD for pipelines, model versioning, environment promotion, and infrastructure-as-code.\n\n- Optimize performance and cost across compute clusters, jobs, and storage layers.\n\n- Implement and manage the enterprise data catalog, including schema design, table ownership, lineage, governance, and documentation using Unity Catalog.\n\nTech Stack & Requirements : \n\n- Databricks platform experience\n\n- Python development for data processing and ETL pipelines\n\n- Unity Catalog knowledge\n\n- AWS data services (S3, IAM, VPC, potentially Glue/Lambda)\n\n- Data lake/lakehouse architecture patterns\n\n- Dashboard building experience\n\nNice to Have : \n\n- RESTful API design and development (Flask, FastAPI, or similar)\n\n- Authentication/authorization patterns (OAuth, API keys, IAM roles)\n\n- Query optimization and performance tuning\n\n- PySpark optimization experience\n\n- ML/AI pipeline experience\n\n- Databricks AI/BI\n\nSkills\nData Engineering, Databricks, Python, Data Pipeline, ETL, Data Ingestion, AWS, Data Catalogue, DataLake","description_format":"text","description_chars":1749,"description_truncated":false,"requirements":{"experience_years_min":7,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":null,"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-09-25T14:00:00Z"}],"liveness":{"score":70,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.775,"p_room":0.9,"age_days":12,"expected_fill_days":23,"reasons":["seen:12","velocity","win:mid"],"computed_at":"2026-09-30T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/codvo-data-engineer","json_url":"https://alion.io/job/codvo-data-engineer.json","meta":{"generated_at":"2026-10-01T02:04:43Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":1705,"day_limit":5000,"remaining_today":3295,"minute_limit":60,"resets_at":"2026-10-02T00:00:00Z"}}}