[{"data":1,"prerenderedAt":272},["ShallowReactive",2],{"story":3},{"name":4,"created_at":5,"published_at":6,"updated_at":7,"id":8,"uuid":9,"content":10,"slug":263,"full_slug":264,"sort_by_date":25,"position":265,"tag_list":266,"is_startpage":166,"parent_id":267,"meta_data":25,"group_id":268,"first_published_at":269,"release_id":25,"lang":270,"path":25,"alternates":271,"default_full_slug":25,"translated_slugs":25},"AI-INFRA 100-02: AI Infrastructure","2026-09-18T10:50:06.993Z","2026-09-18T11:06:12.267Z","2026-09-18T11:06:29.491Z",221343772670568,"6c7fd08a-1eb4-4a2d-9fa2-ee141468850d",{"seo":11,"_uid":15,"type":16,"intro":17,"title":18,"duration":13,"component":19,"course_id":20,"technology":21,"description":22,"on_schedule":166,"course_level":167,"hide_sidebar":168,"prerequisites":169,"on_demand_link":172,"lab_requirements":175,"course_objectives":215,"follow_up_courses":218,"who_should_attend":221,"additional_content":250,"on_demand_training":166,"section_below_hero":255,"hide_class_schedule_cta":168,"hide_private_training_cta":166,"section_below_hero_bg_color":13},{"_uid":12,"title":13,"plugin":14,"og_image":13,"og_title":13,"description":13,"twitter_image":13,"twitter_title":13,"og_description":13,"twitter_description":13},"867ef41d-665c-4afb-ab8c-7929189d5cb2","","seo_metatags","8fdb6453-e42e-443a-befd-e555c41d99b5","course","A four-course series for engineers building and operating GPU clusters","(AI-INFRA 100-02)","training_course","AI Infrastructure","cn",{"type":23,"attrs":24,"content":26},"doc",{"backgroundColor":25},null,[27,121,128],{"type":28,"attrs":29,"content":31},"ordered_list",{"order":30},1,[32,58,79,100],{"type":33,"content":34},"list_item",[35,45,50],{"type":36,"attrs":37,"content":38},"paragraph",{"textAlign":25},[39],{"text":40,"type":41,"marks":42},"AI Infrastructure Foundations","text",[43],{"type":44},"bold",{"type":36,"attrs":46,"content":47},{"textAlign":25},[48],{"text":49,"type":41},"A conceptual walkthrough of the stack: the GPU node, HBM and the checkpoint burst, NVLink, the four fabrics, the management plane, and the failure domains at each layer. The vocabulary and architecture the other three courses assume.",{"type":36,"attrs":51,"content":52},{"textAlign":25},[53],{"text":54,"type":41,"marks":55},"90 minutes, no lab",[56],{"type":57},"italic",{"type":33,"content":59},[60,67,72],{"type":36,"attrs":61,"content":62},{"textAlign":25},[63],{"text":64,"type":41,"marks":65},"Redfish for Fleet Operations",[66],{"type":44},{"type":36,"attrs":68,"content":69},{"textAlign":25},[70],{"text":71,"type":41},"Hardware management and observability through one HTTPS API across a mixed-vendor fleet: health validation before provisioning, firmware compliance and drift remediation, the scrape pipeline behind a fleet dashboard, and VirtualMedia provisioning boot.",{"type":36,"attrs":73,"content":74},{"textAlign":25},[75],{"text":76,"type":41,"marks":77},"2 hours, 7 labs",[78],{"type":57},{"type":33,"content":80},[81,88,93],{"type":36,"attrs":82,"content":83},{"textAlign":25},[84],{"text":85,"type":41,"marks":86},"GPU Cluster Networking",[87],{"type":44},{"type":36,"attrs":89,"content":90},{"textAlign":25},[91],{"text":92,"type":41},"RoCE v2 lossless fabric: why RDMA cannot tolerate loss, PFC and ECN on both switch and NIC, fabric telemetry with NetQ, and two live incidents diagnosed from symptom to root cause without the fault class given in advance.",{"type":36,"attrs":94,"content":95},{"textAlign":25},[96],{"text":97,"type":41,"marks":98},"2 hours, 4 labs plus one optional",[99],{"type":57},{"type":33,"content":101},[102,109,114],{"type":36,"attrs":103,"content":104},{"textAlign":25},[105],{"text":106,"type":41,"marks":107},"AI Cluster Provisioning",[108],{"type":44},{"type":36,"attrs":110,"content":111},{"textAlign":25},[112],{"text":113,"type":41},"Bare-metal servers to a cluster ready for ML-team handoff: host enrollment, declarative provisioning with k0rdent and Metal3/Ironic, the",{"type":36,"attrs":115,"content":116},{"textAlign":25},[117],{"text":118,"type":41,"marks":119},"GPU Operator and Network",[120],{"type":57},{"type":36,"attrs":122,"content":123},{"textAlign":25},[124],{"text":125,"type":41,"marks":126},"FORMAT",[127],{"type":44},{"type":129,"content":130},"bullet_list",[131,138,145,152,159],{"type":33,"content":132},[133],{"type":36,"attrs":134,"content":135},{"textAlign":25},[136],{"text":137,"type":41},"Instructor-led",{"type":33,"content":139},[140],{"type":36,"attrs":141,"content":142},{"textAlign":25},[143],{"text":144,"type":41},"Each course stands alone",{"type":33,"content":146},[147],{"type":36,"attrs":148,"content":149},{"textAlign":25},[150],{"text":151,"type":41},"Foundations recommended first",{"type":33,"content":153},[154],{"type":36,"attrs":155,"content":156},{"textAlign":25},[157],{"text":158,"type":41},"Hands-on labs in the three technology courses",{"type":33,"content":160},[161],{"type":36,"attrs":162,"content":163},{"textAlign":25},[164],{"text":165,"type":41},"Lab environments hosted and preconfigured",false,"essentials",true,{"type":23,"content":170},[171],{"type":36},{"id":13,"url":13,"linktype":173,"fieldtype":174,"cached_url":13},"story","multilink",{"type":23,"attrs":176,"content":177},{"backgroundColor":25},[178],{"type":129,"content":179},[180,187,194,201,208],{"type":33,"content":181},[182],{"type":36,"attrs":183,"content":184},{"textAlign":25},[185],{"text":186,"type":41},"Foundations: general datacenter literacy, basic IP and TCP vocabulary",{"type":33,"content":188},[189],{"type":36,"attrs":190,"content":191},{"textAlign":25},[192],{"text":193,"type":41},"Technology courses: Linux terminal, reading JSON and YAML, curl",{"type":33,"content":195},[196],{"type":36,"attrs":197,"content":198},{"textAlign":25},[199],{"text":200,"type":41},"Networking: TCP/IP, VLANs, L2 and L3 switching",{"type":33,"content":202},[203],{"type":36,"attrs":204,"content":205},{"textAlign":25},[206],{"text":207,"type":41},"Provisioning: Kubernetes and kubectl basics",{"type":33,"content":209},[210],{"type":36,"attrs":211,"content":212},{"textAlign":25},[213],{"text":214,"type":41},"No RDMA, DPU, or Redfish experience assumed",{"type":23,"content":216},[217],{"type":36},{"type":23,"content":219},[220],{"type":36},{"type":23,"attrs":222,"content":223},{"backgroundColor":25},[224,231,236,243,248],{"type":36,"attrs":225,"content":226},{"textAlign":25},[227],{"text":228,"type":41,"marks":229},"INFRASTRUCTURE ENGINEER",[230],{"type":44},{"type":36,"attrs":232,"content":233},{"textAlign":25},[234],{"text":235,"type":41},"You have managed servers, switches, and racks your whole career and are now standing up GPU clusters with DPUs, lossless fabric, and declarative provisioning. You need the vocabulary and then the hands-on reps, in that order.",{"type":36,"attrs":237,"content":238},{"textAlign":25},[239],{"text":240,"type":41,"marks":241},"FIELD, PRODUCT, AND OPERATIONS ROLES",[242],{"type":44},{"type":36,"attrs":244,"content":245},{"textAlign":25},[246],{"text":247,"type":41},"You need to hold a credible technical conversation about GPU topology, RDMA fabric, and cluster automation without implementing it. Foundations is built for exactly that, and it assumes no Linux or scripting background.",{"type":36,"attrs":249},{"textAlign":25},{"type":23,"attrs":251,"content":252},{"backgroundColor":25},[253],{"type":36,"attrs":254},{"textAlign":25},{"type":23,"attrs":256,"content":257},{"backgroundColor":25},[258],{"type":36,"attrs":259,"content":260},{"textAlign":25},[261],{"text":262,"type":41},"Four courses covering the AI cluster stack, from architecture vocabulary through hardware management, fabric, and provisioning. Foundations is conceptual and establishes the shared vocabulary the other three assume. The three technology courses are hands-on and independent of each other: take them in any order, or take only the one that matches the work in front of you.","ai-infrastructure","training/courses/ai-infrastructure",-310,[],111882102,"f3cb621b-daf4-4212-8804-864070ba2ca5","2026-09-18T10:56:19.925Z","default",[],1789730290007]