{"id":1147926,"url":"https://alion.io/job/swingtech-consulting-aiml-data-engineer-hybrid-us-citizens-green-cards-local-to-dmv-only","title":"AI/ML Data Engineer- Hybrid (US Citizens/ Green Cards- Local to DMV only)","company":{"id":688029,"name":"Swingtech Consulting","domain":"swingtech.com","url":"https://alion.io/company/swingtech","size_band":null,"is_staffing_agency":false,"is_intermediary":false,"listed_via":null,"ats_vendor":"Rippling","truth_index":null},"role":"Data Science","role_family":"Data Science","seniority":"middle","employment_type":"full_time","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Washington, United States"],"countries":["US"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":98000,"max_usd":185000,"period":"year","method":"role_seniority_country_remote_cell","sample_n":221},"experience_years_min":4,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Agile","optional":false},{"name":"AWS","optional":false},{"name":"Azure","optional":false},{"name":"Embeddings","optional":false},{"name":"ETL/ELT","optional":false},{"name":"GCP","optional":false},{"name":"Hybrid Search","optional":false},{"name":"Least Privilege","optional":false},{"name":"OCR","optional":false},{"name":"Python","optional":false},{"name":"RAG","optional":false},{"name":"Reranking","optional":false},{"name":"SQL","optional":false},{"name":"Amazon Redshift","optional":true},{"name":"Amazon S3","optional":true},{"name":"AWS Glue","optional":true},{"name":"Azure Data Factory","optional":true},{"name":"BigQuery","optional":true},{"name":"Chroma","optional":true},{"name":"Databricks","optional":true},{"name":"FAISS","optional":true},{"name":"FedRAMP","optional":true},{"name":"Google BigQuery","optional":true},{"name":"Milvus","optional":true},{"name":"NIST 800-171","optional":true},{"name":"NIST 800-53","optional":true},{"name":"OpenSearch","optional":true},{"name":"pgvector","optional":true},{"name":"Pinecone","optional":true},{"name":"PostgreSQL","optional":true},{"name":"Snowflake","optional":true},{"name":"Weaviate","optional":true}],"status":"live","first_seen_at":"2026-09-23T15:09:41Z","employer_posted_date":"2026-09-23","last_verified_at":"2026-09-24T16:57:01Z","board_verified":true,"closed_at":null,"days_open":1,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":1},"description":"About Swingtech\nSwingtech delivers innovative Information Technology and Professional Support services to a diverse range of clients across the federal and intelligence communities. With over 15 years of trusted experience as a systems integrator, we apply agile methodologies and deep industry insight to help our customers achieve greater efficiency, compliance, and cost savings. At Swingtech, we’re committed to excellence and long-term success for our clients and our team.\nPosition Summary\nThe AI/ML Data Engineer develops and sustains the secure data pipelines, data products, retrieval foundations, and governance controls that enable DOL’s AI/ML solutions. The position supports structured, semi-structured, and unstructured data sources used for analytics, AI/ML development, document intelligence, RAG, and production AI applications.\nEssential Duties\nDesign, build, test, deploy, and maintain scalable data pipelines for batch, streaming, near-real-time, and event-driven workloads.\nIntegrate approved agency data sources, APIs, file stores, document repositories, relational databases, data lakes, data warehouses, and authorized external sources.\nDevelop ETL/ELT pipelines for data extraction, validation, transformation, normalization, enrichment, de-identification, metadata management, and loading.\nImplement document-ingestion pipelines that support OCR, parsing, classification, metadata extraction, PII detection/redaction, chunking, embeddings, vector indexing, and retrieval workflows.\nCreate and maintain data models, schemas, data dictionaries, metadata structures, catalog records, and data-quality controls.\nImplement data lineage, source provenance, dataset versioning, retention, access controls, and auditability for training, validation, evaluation, and production datasets.\nPreserve the separation of training, validation, and final evaluation datasets through controlled access, versioning, and documented lifecycle processes.\nDevelop and monitor data-quality measures, including completeness, accuracy, timeliness, duplication, validity, freshness, distribution drift, and labeling quality.\nApply data minimization, masking, encryption, access controls, de-identification, and least-privilege safeguards to PII, CUI, and other protected DOL data.\nCollaborate with AI/ML Engineers to optimize retrieval quality, embeddings, vector stores, hybrid search, reranking, citation traceability, and knowledge-base refresh processes.\nDevelop data-pipeline runbooks, technical documentation, source inventories, lineage artifacts, data-quality reports, and operational support procedures.\nSupport security, privacy, ATO, Responsible AI, incident response, MLOps, monitoring, and release-readiness activities.\nRequired Qualifications\nBachelor’s degree in computer science, data engineering, data science, information systems, software engineering, mathematics, or a related technical discipline.\nAt least four years of experience in data engineering, database development, analytics engineering, ETL/ELT development, data-platform implementation, or related work.\nStrong SQL and Python development skills.\nExperience designing data pipelines and integrating APIs, databases, file systems, cloud storage, data warehouses, or data lakes.\nExperience with data modeling, metadata, data quality, data lineage, data transformation, monitoring, and operational support.\nFamiliarity with AWS, Azure, Google Cloud, or equivalent cloud data services.\nKnowledge of secure data-handling practices, including access control, encryption, data masking, PII protection, and logging.\nMust be willing to work 3 days onsite at customer site in Washington, DC.\nPreferred Qualifications\nExperience with AWS Glue, S3, Athena, Redshift, Lake Formation, Azure Data Factory, Azure Data Lake Storage, Databricks, Snowflake, BigQuery, or equivalent platforms.\nExperience with vector databases or vector-search capabilities, including OpenSearch, pgvector, Pinecone, Weaviate, Milvus, Chroma, FAISS, or similar tools.\nExperience with RAG, document intelligence, OCR, enterprise search, knowledge management, document classification, or content-ingestion pipelines.\nFamiliarity with Federal data governance, FedRAMP, FISMA, NIST 800-53, NIST 800-171, CUI, Privacy Act, and records-management requirements.\nSummary of Benefits\n15 PTO days\n11 paid holidays\nMedical Insurance with - 3 options (HSA with $600 Employer Contribution).\nDental Insurance with no age limit orthodonture.\nVision Insurance through EyeMed in and out of network coverage.\nShort Term and Long-Term Disability coverage with 100% premium support,\nLife insurance and AD&D with 100% premium support\nSupplemental Life Insurance\nCritical Care and Accident Insurance availability\nPet Insurance through Nationwide\nEmployee Assistance Program\n401k with enrollment from day one. 4% deferral by company.\n$1500 Annual Training Budget\n$1500 Referral bonus\nEligibility for annual merit and discretionary bonus\nFlexible work arrangements\nEqual Opportunity Employer Minority/Female/Veterans/Disabled","description_format":"text","description_chars":5045,"description_truncated":false,"requirements":{"experience_years_min":4,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":false},"security_clearance":false,"languages":[]},"benefits":["401k plan","Dental insurance","Flexible schedule","Health insurance","Life insurance","Vision insurance"],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Information Technology","Professional Services","Recruiting & Staffing"],"lifecycle":[{"event":"open","at":"2026-09-23T16:17:21Z"}],"liveness":{"score":86,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.86,"p_room":1,"age_days":0,"expected_fill_days":17,"reasons":["conf:2","win:early"],"computed_at":"2026-09-24T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/swingtech-consulting-aiml-data-engineer-hybrid-us-citizens-green-cards-local-to-dmv-only","json_url":"https://alion.io/job/swingtech-consulting-aiml-data-engineer-hybrid-us-citizens-green-cards-local-to-dmv-only.json","meta":{"generated_at":"2026-09-24T19:27:31Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers"}}