{"id":1303372,"url":"https://alion.io/job/sourceability-principal-nlp-scientist","title":"Principal NLP Scientist","company":{"id":1880443,"name":"Sourceability","domain":"sourceability.com","url":"https://alion.io/company/sourceability","size_band":"201-500","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Greenhouse","truth_index":{"grade":"B","score":75,"open_postings":26,"ghost_share":0,"stale_share":1,"repost_share":0,"time_to_fill_p50_days":null,"computed_at":"2026-09-28T05:45:00Z"}},"role":"AI/ML","role_family":"AI/ML","seniority":"lead","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"inferred","locations":[],"countries":[],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":{"min_usd":72000,"max_usd":171000,"period":"year","method":"global_role_cell_scaled_by_country","sample_n":773},"experience_years_min":8,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Agile","optional":false},{"name":"AWS","optional":false},{"name":"Azure","optional":false},{"name":"Embeddings","optional":false},{"name":"Fine-tuning","optional":false},{"name":"GraphRAG","optional":false},{"name":"Hugging Face","optional":false},{"name":"Knowledge Graph","optional":false},{"name":"LLM","optional":false},{"name":"Machine Learning","optional":false},{"name":"Matplotlib","optional":false},{"name":"NER","optional":false},{"name":"NLP","optional":false},{"name":"NumPy","optional":false},{"name":"ONNX","optional":false},{"name":"Pandas","optional":false},{"name":"Python","optional":false},{"name":"PyTorch","optional":false},{"name":"RAG","optional":false},{"name":"Scikit-learn","optional":false},{"name":"SciPy","optional":false},{"name":"Semantic Search","optional":false},{"name":"Semantic Search","optional":false},{"name":"SQL","optional":false},{"name":"Transformers","optional":false},{"name":"ASP.NET Core","optional":true},{"name":"C#","optional":true},{"name":"CI/CD","optional":true},{"name":"Docker","optional":true},{"name":"Neo4j","optional":true}],"status":"live","first_seen_at":"2026-07-16T16:09:45Z","employer_posted_date":"2026-08-18","last_verified_at":"2026-09-28T20:32:17Z","board_verified":true,"closed_at":null,"days_open":74,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":74},"description":"Sourceability® is a global digital distributor of electronic components transforming how modern businesses bring products to market. With innovation, qualityandlogisticsas the backbone of the company, Sourceability’scutting-edgeproducts and services expeditethe procurement process across a wide range of industries, including communications/cellular, consumer electronics, and auto manufacturing. \nThe Principal NLP Scientist is a senior technical leader responsible for designing, researching, and improving advanced Natural Language Processing and Large Language Model capabilities for production business systems.\nThis role combines applied research, hands-on model development, technical architecture, and practical product impact. The Principal NLP Scientist will lead the design of NLP solutions for named entity recognition, text classification, text generation, semantic search, information extraction, and other language-driven automation use cases.\nThis is not only a research role. The focus is to take modern NLP and LLM technologies and make them reliable, measurable, maintainable, and useful inside real production workflows.\nAssigned Product Group\nProduct Group | NLP / AI Automation\nStream | Software Engineering / AI & Machine Learning\nRole Type | Principal-level individual contributor / technical leader\nThe Principal NLP Scientist will work closely with software engineers, data engineers, product managers, analysts, and data annotation teams to define, build, evaluate, and continuously improve NLP models and language-based automation systems.\nProduct Group Focus Areas\nThe NLP product group is responsible for building and improving systems related to:\nNamed entity recognition and structured data extraction\nText classification and categorization\nText generation and language-based automation\nLarge Language Model evaluation, adaptation, and integration\nRetrieval-augmented generation and semantic search\nKnowledge graph and GraphRAG-based approaches for connecting structured business data, unstructured text, and entity relationships in AI assistant workflows\nData preparation, annotation strategy, and labeling quality\nModel evaluation, monitoring, and production performance\nApplied NLP research and prototype development\nIntegration of NLP models into internal business applications\nInsight on Your Impact\nIn this role, you will influence how the company uses modern NLP and LLM technologies across internal platforms and operational workflows.\nYou will define technical direction for NLP systems, evaluate new approaches, design experiments, create prototypes, and help move successful models into production. Your work will directly affect automation quality, data processing accuracy, operational efficiency, and the long-term AI capabilities of the company.\nThe role requires strong scientific depth, but also practical engineering judgment. The right candidate should be able to read research papers, understand model architecture, design measurable experiments, and also work with engineers to make sure the final solution can run reliably in production.\nYour Qualifications, Your Influence\nTo be successful in this role, you should have:\nPhD in Computer Science, Machine Learning, Artificial Intelligence, Computational Linguistics, Applied Mathematics, Data Science, or a closely related technical field\n8+ years of professional experience in machine learning, artificial intelligence, or NLP\n5+ years of hands-on experience building NLP models for production or near-production systems\nDeep understanding of modern neural network architectures, including RNN, CNN, Transformer-based architectures, attention mechanisms, embeddings, fine-tuning strategies, layers, modules, and loss functions\nStrong practical experience with NLP tasks such as NER, classification, text generation, semantic similarity, information extraction, and document understanding\nStrong experience with Large Language Models, including model evaluation, prompt design, fine-tuning, retrieval-augmented generation, and safe production usage\nPractical understanding of RAG, GraphRAG, knowledge graphs, embeddings, and hybrid retrieval approaches for production LLM applications\nStrong hands-on experience with Python\nStrong experience with PyTorch and Hugging Face Transformers\nExperience with ONNX or other model optimization / model serving formats\nStrong understanding of data preparation, data quality, labeling workflows, annotation guidelines, and model evaluation metrics\nPractical experience with main data analysis and machine learning libraries, including Pandas, NumPy, SciPy, scikit-learn, and Matplotlib\nExperience working with SQL databases and structured business data\nExperience with cloud platforms such as Microsoft Azure or AWS\nAbility to design experiments, define success metrics, compare model approaches, and explain trade-offs clearly\nStrong written and verbal English communication skills\nExperience working in Agile engineering environments\nAbility to provide technical leadership without requiring formal people management authority\nPreferred Skills and Technical Familiarity\nThe following experience will be helpful:\nExperience leading NLP or AI research initiatives in a commercial production environment\nExperience with multilingual NLP systems\nExperience with vector databases, embeddings, semantic search, and RAG architectures\nExperience with knowledge graph concepts, including entity and relationship modeling, graph schema design, traversal queries, and LLM integration with graph databases such as Neo4j, FalkorDB, or similar technologies\nExperience with model serving, monitoring, drift detection, and production ML observability\nExperience with Docker and containerized ML workloads\nExperience with MLOps practices and CI/CD for machine learning systems\nExperience working with data annotation teams and creating annotation instructions\nExperience with .NET / C#, ASP.NET Core, or integration of ML services into enterprise software platforms\nExperience building prototypes, demos, and proof-of-concept applications for new AI capabilities\nPublications, patents, or recognized technical contributions in NLP, machine learning, or applied AI are a plus\nSuccess in the First 90 Days\nDuring the first 90 days, the Principal NLP Scientist is expected to:\nUnderstand the current NLP and AI automation landscape inside the company\nReview existing models, datasets, annotation processes, and production use cases\nIdentify the highest-impact opportunities for NLP and LLM improvements\nDefine practical evaluation metrics for current and future NLP models\nCreate a technical roadmap for improving NER, classification, generation, and information extraction capabilities\nPropose clear standards for data labeling quality, model validation, and production readiness\nDeliver at least one meaningful prototype or improvement proposal with measurable business value\nEstablish strong working relationships with engineering, product, data, and operations stakeholders\nWhat This Role Does Not Own\nThis role does not own general IT infrastructure, end-user support, business operations, or manual data entry processes.\nThe Principal NLP Scientist is also not the sole owner of product priorities or business requirements. Product management owns business prioritization, backlog structure, and stakeholder alignment. This role owns the scientific and technical direction for NLP and LLM capabilities and provides expert guidance on what is technically possible, reliable, and production-ready.\nEQUAL OPPORTUNITY EMPLOYER. \nIt is our policy to abide by all federal, state and local laws prohibiting employment discrimination based on a person’s race, color, religious creed, sex, national origin, ancestry, citizenship status, pregnancy, childbirth, physical disability, mental and/or intellectual disability, age, military status, veteran status (including protected veterans), marital status, registered domestic partner or civil union status, familial status, gender (including sex stereotyping and gender identity or expression), medical condition (including, but not limited to, cancer related or HIV/AIDS related), genetic information, sexual orientation, or any other protected status.","description_format":"text","description_chars":8224,"description_truncated":false,"requirements":{"experience_years_min":8,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"phd","optional":false},"security_clearance":false,"languages":[{"language":"English","level":"Upper-Intermediate (B2)","optional":false}]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Transportation & Logistics","Supply Chain","Electronic Components Distribution"],"lifecycle":[{"event":"open","at":"2026-09-26T12:39:37Z"}],"liveness":{"score":8,"band":"cold","label":"Long shot","p_open":1,"p_active":0.233,"p_room":0.36,"age_days":73,"expected_fill_days":39,"reasons":["conf:3","stale_co","evergreen","win:tail","crowd:"],"computed_at":"2026-09-28T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/sourceability-principal-nlp-scientist","json_url":"https://alion.io/job/sourceability-principal-nlp-scientist.json","meta":{"generated_at":"2026-09-29T04:42:29Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4462,"day_limit":5000,"remaining_today":538,"minute_limit":60,"resets_at":"2026-09-30T00:00:00Z"}}}