{"id":1551778,"url":"https://alion.io/job/nvidia-software-engineer-intern-ai-and-dl-kernel-libraries-2027","title":"Software Engineer Intern, AI and DL Kernel Libraries - 2027","company":{"id":6,"name":"NVIDIA","domain":"nvidia.com","url":"https://alion.io/company/nvidia","size_band":"5000+","is_staffing_agency":false,"employer_type":"direct","is_intermediary":false,"listed_via":null,"ats_vendor":"Workday","truth_index":{"grade":"A","score":89,"open_postings":297,"ghost_share":0.02,"stale_share":0.367,"repost_share":0.04,"time_to_fill_p50_days":28,"computed_at":"2026-10-01T05:45:00Z"}},"role":"Backend","role_family":"Backend","seniority":"intern","employment_type":"internship","work_mode":"on_site","remote_scope":null,"remote_scope_basis":null,"remote_working_hours":null,"hiring_geo_confidence":"structured","locations":["Shanghai, China"],"countries":["CN"],"hiring_countries":[],"hiring_countries_total":0,"salary":null,"salary_estimate":null,"experience_years_min":null,"visa_sponsorship":false,"relocation_package":false,"has_equity":false,"technologies":[{"name":"Apache TVM","optional":false},{"name":"C++","optional":false},{"name":"Computer Vision","optional":false},{"name":"CUDA","optional":false},{"name":"CUDA Toolkit","optional":false},{"name":"cuDNN","optional":false},{"name":"JAX","optional":false},{"name":"LLM","optional":false},{"name":"Machine Learning","optional":false},{"name":"MLIR","optional":false},{"name":"ONNX","optional":false},{"name":"Python","optional":false},{"name":"PyTorch","optional":false},{"name":"PyTorch C++","optional":false},{"name":"Recommender Systems","optional":false},{"name":"SGLang","optional":false},{"name":"TensorFlow","optional":false},{"name":"TensorFlow C++","optional":false},{"name":"TensorRT-LLM","optional":false},{"name":"vLLM","optional":false},{"name":"TensorRT","optional":true}],"status":"live","first_seen_at":"2026-09-30T00:00:00Z","employer_posted_date":"2026-09-30","last_verified_at":"2026-10-01T18:49:34Z","board_verified":true,"closed_at":null,"days_open":2,"trust":{"level":"ok","repost_count":null,"flags":[],"days_open":2},"description":"NVIDIA is looking for outstanding Software Engineer Interns to help develop groundbreaking technologies for AI and deep learning kernel libraries. Our team builds core software that accelerates high-impact AI workloads on NVIDIA GPUs, with a strong focus on deep learning primitives, kernel libraries, and performance-critical GPU software. As an intern on the team, you will contribute to the design, development, optimization, and delivery of software that powers NVIDIA's AI platform.\nThis internship is centered on foundational library engineering, with opportunities to work on low-level kernels, performance primitives, and efficient implementations for modern AI and deep learning workloads. You may contribute to GPU-accelerated deep learning primitives, attention kernel implementations, runtime components, code generation systems, and other performance-critical infrastructure for large language models and advanced AI applications. You will collaborate with world-class engineers across deep learning software, compilers, GPU architecture, and open-source inference ecosystems, and your work can directly impact the performance of real-world workloads at scale.\nWhat you'll be doing\nContribute to production-quality software that ships as part of NVIDIA's AI software stack, including cuDNN, FlashInfer, and optimized support for large language model inference workloads.\nHelp develop new AI systems technologies for efficient inference, with a focus on performance, scalability, maintainability, and usability.\nSupport the design, implementation, and optimization of kernels for high-impact AI workloads across LLM inference, generative AI, computer vision, autonomous driving, and recommender systems.\nAssist in building extensible software abstractions for deep learning libraries, LLM serving engines, and runtime systems.\nContribute to just-in-time compilation, code generation, and runtime technologies for performance-critical GPU workloads.\nAnalyze workload performance, tune current software, and help propose improvements to future software and hardware-software interfaces.\nCollaborate closely with engineers across deep learning frameworks, libraries, kernels, compilers, and GPU architecture teams at NVIDIA.\nContribute to open-source communities and ecosystem integrations where relevant, including projects such as FlashInfer, vLLM, and SGLang.\nWhat we need to see\nCurrently pursuing a Bachelor's, Master's, or PhD degree in Computer Science, Electrical Engineering, or a related field.\nCoursework, research, or hands-on project experience in machine learning, deep learning systems, compilers, systems software, or GPU programming.\nStrong programming skills in C/C++ and Python.\nFamiliarity with CUDA development and GPU programming fundamentals.\nExperience developing with or using deep learning frameworks such as PyTorch, JAX, TensorFlow, or ONNX.\nUnderstanding of linear algebra, performance analysis, profiling, and code optimization.\nInterest in software abstractions, APIs, and higher-level system architecture for performance-sensitive systems.\nInterest in modern machine learning and inference system trends, especially around LLMs and generative AI.\nStrong problem-solving skills, curiosity, and the ability to work effectively in a collaborative environment.\nWays to stand out from the crowd\nHands-on experience with inference engines and runtimes such as vLLM, SGLang, MLC, TensorRT-LLM, or similar systems.\nBackground in domain-specific compilers, code generation, or library solutions for LLM inference and training.\nExposure to machine learning compilers or IR systems such as MLIR, Apache TVM, TensorIR, or related technologies.\nPractical experience with GPU performance modeling, computer architecture, or accelerator-oriented software design.\nOpen-source project ownership or meaningful contributions in deep learning systems, compilers, kernels, or inference infrastructure.","description_format":"text","description_chars":3921,"description_truncated":false,"requirements":{"experience_years_min":null,"management_years_min":null,"team_size_min":null,"manages_managers":false,"education":{"level":"bachelor","optional":false},"security_clearance":false,"languages":[]},"benefits":[],"hiring_locations":[],"hiring_excludes":[],"relocation_offered":false,"industries":["Processors, MCUs & AI Chips","Servers & Data Center Hardware","Computer Components","AI Chips & Accelerators"],"lifecycle":[{"event":"open","at":"2026-10-01T00:24:56Z"}],"liveness":{"score":90,"band":"hot","label":"Hiring now","p_open":1,"p_active":0.903,"p_room":1,"age_days":1,"expected_fill_days":28,"reasons":["conf:5","velocity","win:early","comp:junior,brand"],"computed_at":"2026-10-01T05:45:00Z"},"pay":null,"html_url":"https://alion.io/job/nvidia-software-engineer-intern-ai-and-dl-kernel-libraries-2027","json_url":"https://alion.io/job/nvidia-software-engineer-intern-ai-and-dl-kernel-libraries-2027.json","meta":{"generated_at":"2026-10-02T03:29:14Z","cache_seconds":300,"methodology":"https://alion.io/methodology","terms":"https://alion.io/terms","contact":"https://alion.io/contact","api":"https://alion.io/developers","usage":{"tier":"crawler","counted_by":"address","units_charged":1,"used_today":4895,"day_limit":5000,"remaining_today":105,"minute_limit":60,"resets_at":"2026-10-03T00:00:00Z"}}}