{"id":1079743,"url":"https://alion.io/job/mydna-senior-data-pipeline-engineerdeveloper","title":"Senior Data Pipeline Engineer/Developer","company":{"id":680223,"name":"MYDNA","domain":"mydna.com","url":"https://alion.io/company/mydna-3","size_band":"51-200","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Rippling","truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"senior","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Houston, United States"],"countries":["US"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":106000,"max_usd":203000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":1079},"experience_years_min":8,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":".NET","optional":false},{"name":"Amazon Kinesis","optional":false},{"name":"Amazon Redshift","optional":false},{"name":"Amazon S3","optional":false},{"name":"Apache Kafka","optional":false},{"name":"Argo Workflows","optional":false},{"name":"AWS","optional":false},{"name":"AWS Lambda","optional":false},{"name":"AWS Step Functions","optional":false},{"name":"C#","optional":false},{"name":"C++","optional":false},{"name":"CI/CD","optional":false},{"name":"CloudFormation","optional":false},{"name":"Dagster","optional":false},{"name":"dbt","optional":false},{"name":"Docker","optional":false},{"name":"ETL/ELT","optional":false},{"name":"GDPR","optional":false},{"name":"HIPAA","optional":false},{"name":"Kubernetes","optional":false},{"name":"Least Privilege","optional":false},{"name":"MS SQL","optional":false},{"name":"PostgreSQL","optional":false},{"name":"Prefect","optional":false},{"name":"Pulumi","optional":false},{"name":"Python","optional":false},{"name":"Snowflake","optional":false},{"name":"Spark","optional":false},{"name":"SQL","optional":false},{"name":"Synthetic Data","optional":false},{"name":"Terraform","optional":false}],"status":"live","first_seen_at":"2026-09-17T16:32:34Z","employer_posted_date":"2026-09-17","last_verified_at":"2026-10-11T00:19:35Z","board_verified":true,"closed_at":null,"days_open":23,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":23},"description":"Our Purpose\nOur mission is to build a healthier and more connected world with precision health and genealogy services. We empower individuals with actionable insights into their genetic makeup, fostering a deeper understanding of their ancestry, health, and wellness. By integrating the experience of Gene by Gene Laboratory Services, FamilyTreeDNA genealogy, and myDNA reporting services, we strive to deliver cutting-edge genetic testing and personalized solutions that inspire informed decisions and enhance quality of life. Our team is dedicated to advancing the field of genomics through innovation, research, and a commitment to excellence.\nOur Values\nAll employees are expected to demonstrate our values of Innovate, One Team, and Integrity when carrying out the accountabilities and responsibilities of their role. This is how we show up every day for ourselves, our colleagues and our customers and strategic partners to deliver our vision and strategic goals.\nPosition Overview\nWe are seeking a Senior Data Pipeline Engineer/Developer to design, build, and own the data pipelines that move genomic and operational data from lab instruments and LIMS through to analytics, products, and clinical reporting. In this senior individual-contributor role you will set technical direction for our data platform, build production pipelines that are reliable and reproducible at scale, and provide architectural and code-level guidance to existing engineering teams. You will own data quality, lineage, and observability end to end, and operate in a regulated environment where reproducibility and auditability are non-negotiable. This is a Python-first, full-stack engineering role. This is a contractor position, slated for a 12-month term.\nAccountabilities and Responsibilities\nPipeline Engineering & Integration\nBuilds and operates production ETL and ELT for high-volume genomic data as well as operational and business data.\nIntegrates data across LIMS, lab instruments, internal applications, and third-party sources through robust, well-tested interfaces.\nDevelops across the stack in Python (data services, internal APIs, and supporting application code) and provides architectural and code-level guidance to engineering teams on data-layer integration.\n\nArchitecture & Technical Leadership\nSets technical direction for batch and streaming data pipelines by evaluating frameworks, orchestration, storage, and processing patterns, making recommendations, and leading adoption.\nModels and tunes data stores (Microsoft SQL Server and PostgreSQL, plus a cloud warehouse or lake) for performance and scale.\nDefines and enforces engineering standards for testing, CI/CD, infrastructure as code, code review, and architecture decision records.\n\nData Quality & Pipeline Observability\nOwns data quality, lineage, and observability, including freshness, completeness, schema-drift detection, cost-per-job, and SLAs.\nBuilds pipelines for reproducibility and audit-readiness, incorporating versioned data and code, lineage, decision logging, access controls, and evidence collection.\nPartners with security and compliance on data privacy, PII and PHI handling, and regulatory requirements across the data lifecycle.\n\nRegulatory Compliance and Security Governance\nApply deep understanding of CAP/CLIA, HIPAA, GDPR, and GxP regulations specifically to data pipeline architecture.\nOversee protected health information (PHI) handling, data lineage, retention, and comprehensive audit logging.\nProduce and maintain the critical operational evidence required to carry the data platform successfully through compliance audits.\nAdhere to strict data-governance controls necessary for securely handling sensitive genomic data across global regions.\nEnforce least-privilege access and ensure zero data egress to personal or unapproved infrastructure.\nUtilize exclusively de-identified or synthetic data within development environments.\nMaintain strict compliance with international data-residency and localization requirements.\n\nPosition Requirements\nSkills and Knowledge\nStrong SQL proficiency on Microsoft SQL Server and PostgreSQL, including schema design, query tuning, and performance troubleshooting.\nStrong Python proficiency across the stack (data pipelines, backend services, and APIs). This is a Python-first role.\nProduction experience with a workflow orchestrator (Argo Workflows, Step Functions, Prefect, Dagster, Airflow, or similar).\nProduction experience with containers (Docker) and Kubernetes, which the lead orchestrator (Argo Workflows) runs on.\nStrong testing discipline, including unit, integration, and data-quality or contract tests.\nExcellent written and verbal communication, including the ability to explain data tradeoffs to technical and compliance stakeholders.\nGenomics or NGS data formats and handling (FASTQ, BAM/CRAM, VCF).\nBioinformatics workflow engines (Nextflow, WDL/Cromwell, or Snakemake).\nAWS data stack proficiency (S3, Glue, EMR, Batch, Lambda, Redshift) and/or AWS HealthOmics.\nDistributed processing and streaming frameworks (Spark, Kafka, Kinesis).\nData warehouse and transformation tooling (Snowflake, dbt).\nInfrastructure as code practices (Terraform, CDK, CloudFormation, or Pulumi).\nFamiliarity with .NET (C#) and/or C++ as supporting languages for integrating with existing services.\nKnowledge of LIMS integration and on-premises plus hybrid data architectures.\n\nExperience\n8+ years of professional software or data engineering experience.\n4+ years building and operating production data pipelines at scale.\n3+ years of production cloud experience (AWS preferred).\nDemonstrable experience working in regulated environments. Compliance is a hard requirement for this role.\nExperience supporting HIPAA or GxP audits.\n\nEducation\nBachelor's degree in Computer Science, Bioinformatics, or a related field, or equivalent professional experience.\nAn advanced degree in a quantitative field is preferred. \n\nWhy Join Us\nAt Gene by Gene, you’ll join a mission-driven team advancing the science of genetics and discovery. You’ll have the opportunity to shape meaningful campaigns, tell compelling brand stories, and collaborate with talented professionals who share your passion for creativity, curiosity, and impact.","description_format":"text","description_chars":6243,"description_truncated":false,"requirements":{"experience_years_min":8,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":false},"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Genomics & Genetic Testing","Precision Medicine"],"lifecycle":[{"event":"open","at":"2026-09-20T16:33:49Z"}],"visa":[],"liveness":{"score":68,"band":"ok","label":"Likely open","p_open":1,"p_active":0.755,"p_room":0.9,"age_days":22,"expected_fill_days":39,"reasons":["conf:1","win:mid"],"computed_at":"2026-10-10T05:45:15Z"},"pay":null,"html_url":"https://alion.io/job/mydna-senior-data-pipeline-engineerdeveloper","json_url":"https://alion.io/job/mydna-senior-data-pipeline-engineerdeveloper.json","meta":{"generated_at":"2026-10-11T03:02:39Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","about":"Alion is a live layer of people, companies and AI agents: who they are, whether they are real and active right now, what they do and how to work with them, readable by people and by agents and paid per call.","catalog":"https://alion.io/catalog.json","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":302,"day_limit":5000,"remaining_today":4698,"minute_limit":60,"resets_at":"2026-10-12T00:00:00Z"}},"offers":[{"id":"company.slices","title":"One company in depth, by slice","status":"live","price":{"credits":0.02,"usd":0.002,"plus_per_slice":{"credits":0.05,"usd":0.005}},"unit":"per company, plus each slice with data","note":"the employer in depth","call":{"mcp_tool":"get_company","arguments":{"id":680223},"rest":"https://alion.io/mcp/rest/get_company?id=680223"},"human":"https://alion.io/catalog?offer=company.slices&for=job%2Fmydna-senior-data-pipeline-engineerdeveloper"},{"id":"market.stats","title":"A market slice: pay, demand and time to fill","status":"live","price":{"credits":1,"usd":0.1},"unit":"per slice","note":"pay, demand and time to fill for this role and place","call":{"mcp_tool":"market_stats"},"human":"https://alion.io/catalog?offer=market.stats&for=job%2Fmydna-senior-data-pipeline-engineerdeveloper"},{"id":"job.search","title":"Open jobs by role, technology, place, pay and visa","status":"live","price":{"credits":0.02,"usd":0.002},"unit":"per posting in a list","note":"similar open postings","call":{"mcp_tool":"search_jobs"},"human":"https://alion.io/catalog?offer=job.search&for=job%2Fmydna-senior-data-pipeline-engineerdeveloper"},{"id":"company.verify","title":"Is this company real and active right now","status":"pilot","price":null,"unit":"per company","request":{"url":"https://alion.io/catalog/request","method":"POST","body":"{\"offer\": \"company.verify\", \"for\": \"job/mydna-senior-data-pipeline-engineerdeveloper\", \"note\": \"what you need it for\"}"},"human":"https://alion.io/catalog?offer=company.verify&for=job%2Fmydna-senior-data-pipeline-engineerdeveloper"}]}