{"id":1650642,"url":"https://alion.io/job/nmrk-staff-data-engineer","title":"Staff Data Engineer","company":{"id":1824102,"name":"Newmark","domain":"nmrk.com","url":"https://alion.io/company/nmrk","size_band":"501-1000","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Oracle","truth_index":{"grade":"B","score":75,"open_postings":28,"ghost_share":0,"stale_share":1,"repost_share":0,"time_to_fill_p50_days":null,"computed_at":"2026-10-06T05:45:30Z"}},"role":"Data Science","role_family":"Data Science","seniority":"staff","employment_type":null,"work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Chicago, United States"],"countries":["US"],"hiring_countries":[],"hiring_countries_total":0,"salary":{"min":190000,"max":225000,"currency":"USD","period":"year","gross":null,"usd_annual":225000},"salary_estimate":null,"experience_years_min":12,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"AI Agents","optional":false},{"name":"Amazon Redshift","optional":false},{"name":"AutoGen","optional":false},{"name":"AWS","optional":false},{"name":"Azure","optional":false},{"name":"BigQuery","optional":false},{"name":"CI/CD","optional":false},{"name":"Data Vault","optional":false},{"name":"Databricks","optional":false},{"name":"Embeddings","optional":false},{"name":"ETL/ELT","optional":false},{"name":"Feature Store","optional":false},{"name":"GCP","optional":false},{"name":"Google BigQuery","optional":false},{"name":"GraphQL","optional":false},{"name":"Informatica","optional":false},{"name":"Java","optional":false},{"name":"LangChain","optional":false},{"name":"LLM","optional":false},{"name":"LLMOps","optional":false},{"name":"Machine Learning","optional":false},{"name":"Master Data Management","optional":false},{"name":"Model Context Protocol","optional":false},{"name":"Python","optional":false},{"name":"RAG","optional":false},{"name":"Rest API","optional":false},{"name":"Scala","optional":false},{"name":"Semantic Kernel","optional":false},{"name":"Snowflake","optional":false},{"name":"SQL","optional":false},{"name":"Alation","optional":true},{"name":"Amazon Kinesis","optional":true},{"name":"Apache Kafka","optional":true},{"name":"Collibra","optional":true},{"name":"Dagster","optional":true},{"name":"dbt","optional":true},{"name":"Docker","optional":true},{"name":"Flink","optional":true},{"name":"Kubernetes","optional":true},{"name":"Multi-Agent Systems","optional":true},{"name":"Spark","optional":true},{"name":"Tool Use","optional":true}],"status":"live","first_seen_at":"2026-09-24T16:54:42Z","employer_posted_date":"2026-09-24","last_verified_at":"2026-10-07T01:04:52Z","board_verified":true,"closed_at":null,"days_open":12,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":12},"description":"Own and drive the technical architecture for complex, cross-team data initiatives spanning ingestion, transformation, storage, and serving layers.\n\nDesign, build, and maintain scalable, high-performance data pipelines and distributed data platforms in a cloud-native environment (Azure, AWS, or GCP).\n\nArchitect and lead enterprise Master Data Management (MDM), including golden records, entity resolution, data domains, reference and hierarchy management, and stewardship, to create trusted, authoritative data across the business.\n\nArchitect and integrate agentic AI and LLM-driven workflows (autonomous agents, RAG pipelines, AI copilots) into data platforms and pipelines to drive efficiency and new capabilities.\n\nBuild and support the data foundations for machine learning and AI, including feature stores, vector stores, embeddings, and ML/LLMOps pipelines.\n\nDesign and deliver backend data services and APIs (REST/GraphQL), and contribute across the stack to expose curated datasets to applications, analytics, and BI consumers.\n\nSet engineering standards and best practices for data quality, modeling, testing, observability, and deployment across the organization, including responsible use of AI-assisted development tools.\n\nEstablish data governance, lineage, cataloging, and quality frameworks across the data estate.\n\nLead technical design reviews and provide architectural guidance to multiple engineering and data teams.\n\nPartner with product, analytics, and engineering leadership to translate business strategy into scalable data roadmaps, including AI-driven capabilities.\n\nIdentify and resolve systemic performance, reliability, and scalability issues across the data stack.\n\nMentor and coach senior and mid-level engineers, raising the technical bar across the organization on data engineering, MDM, and AI practices.\n\nDrive adoption of modern frameworks, tools, and engineering practices, including agentic AI and LLM tooling, to improve delivery velocity and platform resilience.\n\nMaintain awareness of emerging technologies and industry trends, particularly in agentic AI, master data management, and modern data platforms, and assess their applicability to the business.\n\n Basic Qualifications\nBachelor's degree in Computer Science, Engineering, MIS, or related field preferred.\n\n12+ years of experience in data engineering or software engineering, with demonstrated experience architecting data platforms and pipelines at scale.\n\nExpert-level SQL and strong proficiency in Python (Scala or Java a plus) for large-scale data processing and transformation.\n\nDeep experience with cloud data platforms (e.g., Databricks, Snowflake, Synapse, BigQuery, Redshift) and cloud-native architecture patterns.\n\nDeep understanding of distributed systems, data modeling (dimensional, data vault, lakehouse), and ETL/ELT architecture.\n\nHands-on experience designing and implementing Master Data Management (MDM) solutions, including entity resolution, match/merge, golden records, and reference/hierarchy management (e.g., Informatica, Reltio, Profisee, or similar).\n\nHands-on experience building or integrating agentic AI systems, LLM-powered applications, RAG pipelines, or AI agent orchestration frameworks (e.g., LangChain, AutoGen, Semantic Kernel, MCP).\n\nExperience building backend data services and APIs (REST/GraphQL), with comfort working across the full stack.\n\nStrong background with both relational (SQL) and NoSQL data stores, plus data lake/lakehouse formats (Delta, Iceberg, Parquet).\n\nDeep understanding of CI/CD pipelines, infrastructure as code, and DevOps/DataOps practices.\n\nProven track record of leading large-scale technical initiatives across multiple teams.\n\nDemonstrated ability to mentor engineers and influence technical direction without direct reporting authority.\n\nPreferred Qualifications\nExperience with data governance, lineage, and cataloging tools (e.g., Unity Catalog, Microsoft Purview, Collibra, Alation).\n\nExperience designing multi-agent systems, tool-calling architectures, or retrieval-augmented generation (RAG) pipelines.\n\nExperience with event-driven architectures and streaming/real-time data processing (e.g., Kafka, Event Hubs, Kinesis, Flink, Spark Structured Streaming).\n\nExperience building the data layer for ML/AI, including feature stores, vector databases, embeddings, and ML/LLMOps.\n\nFamiliarity with containerization and orchestration (Docker, Kubernetes) and workflow orchestration (Airflow, Dagster, dbt).\n\nPrior experience in commercial real estate, fintech, or operations/transaction systems.\n\nTrack record of speaking, writing, or open-source contributions that demonstrate technical thought leadership, especially in applied AI or data.\n\nWhy Join Us?\nShape the technical direction of business-critical data platforms at enterprise scale, including master data management and next-generation agentic AI initiatives.\n\nBe part of a high-impact team where ownership, innovation, and technical excellence drive success.\n\nCompetitive compensation, growth opportunities, and access to world-class engineering, data, and AI resources.\n\nCollaborative Culture: Join a high-caliber team with deep expertise across data engineering, cloud, MDM, agentic AI, and distributed systems.\n\nGrowth & Learning: Access world-class learning resources and mentorship to advance your career.\n\nWork-Life Balance: Flexible working hours and hybrid options.\n\nBenefits: Comprehensive health, dental and vision insurance.\n\nIf you're passionate about architecting scalable data platforms, building trusted master data and agentic AI-driven solutions, and shaping engineering culture, we'd love to hear from you!\nApply now and help redefine the future of data and AI at scale!\nSalary Language:\nThe expected base salary for this position ranges from $190,000 to $225,000 annually. The actual base salary will be determined on an individualized basis taking into account a wide range of factors including, but not limited to, relevant skills, experience, education, and, where applicable, licenses or certifications held. In addition to base salary and a competitive benefits package, this position may be eligible for additional types of compensation including discretionary bonuses and other short- and long-term incentives (e.g., deferred cash, equity, etc.).\nWorking Conditions: Normal working conditions with the absence of disagreeable elements.\nNote: The statements herein are intended to describe the general nature and level of work being performed by employees, and are not to be construed as an exhaustive list of responsibilities, duties, and skills required of personnel so classified.\nNewmark is an Equal Opportunity/Affirmative Action employer. All qualified applicants will receive consideration for employment without regard to race, color, religion, sex including sexual orientation and gender identity, national origin, disability, protected Veteran Status, or any other characteristic protected by applicable federal, state, or local law.","description_format":"text","description_chars":6999,"description_truncated":false,"requirements":{"experience_years_min":12,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":true},"security_clearance":false,"languages":[]},"benefits":["Equity","Flexible schedule","Growth opportunities","Vision insurance"],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":[],"lifecycle":[{"event":"open","at":"2026-10-02T00:14:48Z"}],"visa":[],"liveness":{"score":63,"band":"ok","label":"Likely open","p_open":1,"p_active":0.632,"p_room":1,"age_days":11,"expected_fill_days":49,"reasons":["conf:0","stale_co","velocity","win:early"],"computed_at":"2026-10-06T05:45:30Z"},"pay":{"stated_usd_annual":225000,"is_top_pay":true},"html_url":"https://alion.io/job/nmrk-staff-data-engineer","json_url":"https://alion.io/job/nmrk-staff-data-engineer.json","meta":{"generated_at":"2026-10-07T02:11:56Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","about":"Alion is a live layer of people, companies and AI agents: who they are, whether they are real and active right now, what they do and how to work with them, readable by people and by agents and paid per call.","catalog":"https://alion.io/catalog.json","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":3607,"day_limit":5000,"remaining_today":1393,"minute_limit":60,"resets_at":"2026-10-08T00:00:00Z"}},"offers":[{"id":"company.slices","title":"One company in depth, by slice","status":"live","price":{"credits":0.02,"usd":0.002,"plus_per_slice":{"credits":0.05,"usd":0.005}},"unit":"per company, plus each slice with data","note":"the employer in depth","call":{"mcp_tool":"get_company","arguments":{"id":1824102},"rest":"https://alion.io/mcp/rest/get_company?id=1824102"},"human":"https://alion.io/catalog?offer=company.slices&for=job%2Fnmrk-staff-data-engineer"},{"id":"market.stats","title":"A market slice: pay, demand and time to fill","status":"live","price":{"credits":1,"usd":0.1},"unit":"per slice","note":"pay, demand and time to fill for this role and place","call":{"mcp_tool":"market_stats"},"human":"https://alion.io/catalog?offer=market.stats&for=job%2Fnmrk-staff-data-engineer"},{"id":"job.search","title":"Open jobs by role, technology, place, pay and visa","status":"live","price":{"credits":0.02,"usd":0.002},"unit":"per posting in a list","note":"similar open postings","call":{"mcp_tool":"search_jobs"},"human":"https://alion.io/catalog?offer=job.search&for=job%2Fnmrk-staff-data-engineer"},{"id":"company.verify","title":"Is this company real and active right now","status":"pilot","price":null,"unit":"per company","request":{"url":"https://alion.io/catalog/request","method":"POST","body":"{\"offer\": \"company.verify\", \"for\": \"job/nmrk-staff-data-engineer\", \"note\": \"what you need it for\"}"},"human":"https://alion.io/catalog?offer=company.verify&for=job%2Fnmrk-staff-data-engineer"}]}