AI Cluster Technical Program Manager – Validation, Debug & Agentic AIEmbedded

External Listing
{
  "source": {
    "name": "linkedin",
    "id": "4444501664",
    "url": "https://www.linkedin.com/jobs/view/ai-cluster-technical-program-manager-%E2%80%93-validation-debug-agentic-ai-at-amd-4444501664?_l=en"
  },
  "postedDate": "2026-09-16T17:12:43.357Z",
  "applicationDeadline": null,
  "isActive": true,
  "isExpired": false,
  "matching": {
    "role": {
      "primaryTitle": "Technical Program Manager",
      "titleSynonyms": [
        "AI Cluster Technical Program Manager",
        "Program Manager",
        "TPM"
      ],
      "secondaryTitles": [
        "Validation Manager",
        "Debug Manager",
        "AI Program Manager"
      ],
      "function": "production_specialized_services_managers",
      "functionConfidence": "high",
      "roleFamily": "m132_manufacturing_mining_construction_distribution_managers",
      "roleFamilyConfidence": "high",
      "roleSubFamily": "technical program management",
      "roleSubFamilyConfidence": "high",
      "seniority": "mid_senior",
      "industries": [
        "i112_appliances_electrical_and_electronics_manufacturing"
      ]
    },
    "primarySignals": {
      "tasks": [
        "Define and drive program plans for AI infrastructure systems validation and readiness",
        "Lead execution of GPU-based AI platform bring-up, qualification, and deployment validation",
        "Manage multi-node and multi-rack scale testing and readiness gates",
        "Coordinate debug, failure analysis, and risk management across engineering teams",
        "Lead system rack and cluster-level debug and incident management",
        "Drive development and rollout of AI agents for automated triage and incident management"
      ],
      "skills": [
        "Program management",
        "Hardware and AI infrastructure validation",
        "Debug and failure analysis",
        "Cross-functional team leadership",
        "Risk and issue mitigation",
        "Executive reporting",
        "Data-driven decision making"
      ],
      "tools": [
        "Jira",
        "Confluence",
        "Dashboards",
        "Excel",
        "PowerPoint"
      ],
      "educationLevel": 3,
      "educationKeywords": [
        "Systems Engineering",
        "Electrical Engineering",
        "Computer Science"
      ],
      "certifications": [
        "PMP",
        "Scrum Master"
      ],
      "languages": [],
      "yearsRelevant": 5
    },
    "secondarySignals": {
      "tasks": [
        "Drive post-incident reviews and root-cause analysis",
        "Champion AI-first operational workflows",
        "Partner with data engineering teams for AI-powered operational tools integration"
      ],
      "skills": [
        "GPU cluster scale testing",
        "System stress and performance validation",
        "Infrastructure observability and telemetry pipelines",
        "Incident management frameworks"
      ],
      "tools": [],
      "educationLevel": null,
      "educationKeywords": [],
      "certifications": [],
      "languages": [],
      "yearsRelevant": null
    },
    "practical": {
      "locations": [
        "Austin, TX"
      ],
      "locationProvenance": "stated",
      "countries": [
        "US"
      ],
      "workModes": [
        "on_site"
      ],
      "workModeProvenance": "stated",
      "employmentTypes": [
        "full_time"
      ],
      "compensation": {
        "min": 162640,
        "max": 243960,
        "currency": "USD",
        "period": "year"
      }
    },
    "dataCompleteness": "high"
  },
  "indexing": {
    "function": "production_specialized_services_managers",
    "functionConfidence": "high",
    "roleFamily": "m132_manufacturing_mining_construction_distribution_managers",
    "roleFamilyConfidence": "high",
    "countryCode": "US",
    "countryCodeConfidence": "high",
    "locationBucket": "unknown",
    "locationBucketConfidence": "unknown",
    "workMode": "on_site",
    "workModeConfidence": "high",
    "employmentType": "full_time",
    "employmentTypeConfidence": "high",
    "isAgency": "direct",
    "salaryMax": 179991,
    "industryGroup": "i112_appliances_electrical_and_electronics_manufacturing",
    "industryGroupConfidence": "high",
    "educationRequired": 3,
    "educationRequiredConfidence": "high"
  },
  "display": {
    "title": "AI Cluster Technical Program Manager – Validation, Debug & Agentic AI",
    "company": {
      "name": "AMD"
    },
    "locationDisplay": "Austin, TX",
    "applicationUrl": null
  },
  "roleFamilyEsco": "m133_information_communications_technology_services_managers",
  "requirementsEssentiality": {
    "items": [
      {
        "text": "Experience leading complex hardware or AI infrastructure programs with ownership across bring-up, validation, and deploy",
        "category": "skill",
        "essentiality": "compulsory",
        "triggerPhrase": "Required Qualifications",
        "confidence": "high"
      },
      {
        "text": "Strong technical understanding of GPU-based AI systems, rack architectures, and datacenter infrastructure",
        "category": "skill",
        "essentiality": "compulsory",
        "triggerPhrase": "Required Qualifications",
        "confidence": "high"
      },
      {
        "text": "Strong written and verbal communication skills, including executive-level status reporting",
        "category": "skill",
        "essentiality": "compulsory",
        "triggerPhrase": "Required Qualifications",
        "confidence": "high"
      },
      {
        "text": "Proficiency with program management and execution tools (Jira, Confluence, dashboards, Excel/PowerPoint)",
        "category": "tool",
        "essentiality": "compulsory",
        "triggerPhrase": "Required Qualifications",
        "confidence": "high"
      },
      {
        "text": "Hands-on experience with GPU cluster scale testing, system stress, or performance validation",
        "category": "skill",
        "essentiality": "preferred",
        "triggerPhrase": "Preferred Qualifications",
        "confidence": "high"
      },
      {
        "text": "Familiarity with rack-level bring-up, power/cooling constraints, networking, and failure modes at scale",
        "category": "skill",
        "essentiality": "preferred",
        "triggerPhrase": "Preferred Qualifications",
        "confidence": "high"
      },
      {
        "text": "Experience working through hardware/firmware debug cycles in pre-production or customer-facing environments",
        "category": "skill",
        "essentiality": "preferred",
        "triggerPhrase": "Preferred Qualifications",
        "confidence": "high"
      },
      {
        "text": "Experience managing fleet-scale validation and deployment of AI, cloud, HPC, or hyperscale infrastructure",
        "category": "skill",
        "essentiality": "preferred",
        "triggerPhrase": "Preferred Qualifications",
        "confidence": "high"
      },
      {
        "text": "Strong understanding of infrastructure observability, telemetry pipelines, log analytics, and incident management",
        "category": "skill",
        "essentiality": "preferred",
        "triggerPhrase": "Preferred Qualifications",
        "confidence": "high"
      },
      {
        "text": "Bachelor’s or master’s degree systems, EE, CS, or related engineering discipline",
        "category": "education",
        "essentiality": "compulsory",
        "triggerPhrase": "Academic Credentials",
        "confidence": "high"
      },
      {
        "text": "PMP, Scrum Master, or equivalent program management training",
        "category": "certification",
        "essentiality": "compulsory",
        "triggerPhrase": "Academic Credentials",
        "confidence": "high"
      }
    ]
  }
}