From 5908faaafc91bf7130fba0640784fbaf9f9dae3c Mon Sep 17 00:00:00 2001 From: zeenat28-ui Date: Sat, 19 Sep 2026 11:10:33 -0700 Subject: [PATCH 1/5] Fix issue #385: Decouple Quark quantization via Docker and resolve flexml dependency conflicts --- .../getting_started_resnet/int8/Dockerfile | 42 +++++ .../getting_started_resnet/int8/Readme.md | 158 +++++++++++++++++- .../int8/docker_run.bat | 37 ++++ .../getting_started_resnet/int8/docker_run.sh | 30 ++++ .../int8/prepare_model_data.py | 8 +- .../int8/requirements.txt | 25 ++- .../int8/requirements_infer.txt | 17 ++ .../int8/requirements_quantize.txt | 21 +++ 8 files changed, 329 insertions(+), 9 deletions(-) create mode 100644 CNN-examples/getting_started_resnet/int8/Dockerfile create mode 100644 CNN-examples/getting_started_resnet/int8/docker_run.bat create mode 100644 CNN-examples/getting_started_resnet/int8/docker_run.sh create mode 100644 CNN-examples/getting_started_resnet/int8/requirements_infer.txt create mode 100644 CNN-examples/getting_started_resnet/int8/requirements_quantize.txt diff --git a/CNN-examples/getting_started_resnet/int8/Dockerfile b/CNN-examples/getting_started_resnet/int8/Dockerfile new file mode 100644 index 00000000..20a2dcc3 --- /dev/null +++ b/CNN-examples/getting_started_resnet/int8/Dockerfile @@ -0,0 +1,42 @@ +# ============================================================================== +# Dockerfile: AMD Quark Quantization Pipeline for ResNet INT8 +# MLOps Decoupled Build Environment (Google Cloud Console / Linux / Docker) +# ============================================================================== +# Purpose: +# Isolates the AMD Quark Quantization and PyTorch-to-ONNX export pipeline +# from the host Ryzen AI runtime environment. This completely avoids the +# Python package dependency conflicts with 'flexml 1.8.0' (Issue #385). +# ============================================================================== + +FROM python:3.10-slim + +ENV PYTHONUNBUFFERED=1 \ + DEBIAN_FRONTEND=noninteractive + +WORKDIR /app + +# Install system dependencies +RUN apt-get update && apt-get install -y --no-install-recommends \ + ca-certificates \ + curl \ + wget \ + git \ + libgl1 \ + libglib2.0-0 \ + && rm -rf /var/lib/apt/lists/* + +# Install Python quantization dependencies +COPY requirements_quantize.txt . +RUN pip install --no-cache-dir --upgrade pip && \ + pip install --no-cache-dir -r requirements_quantize.txt + +# Copy source code and model directory +COPY resnet_utils.py prepare_model_data.py resnet_quantize.py ./ +COPY models/ ./models/ + +# Create data directory +RUN mkdir -p /app/data /app/models + +# Entry command: prepare data, export ONNX, and run Quark INT8 quantization +CMD ["sh", "-c", "python prepare_model_data.py && python resnet_quantize.py && echo '=== Quantization Complete! Artifact: models/resnet_quantized.onnx ==='"] + diff --git a/CNN-examples/getting_started_resnet/int8/Readme.md b/CNN-examples/getting_started_resnet/int8/Readme.md index b39b3c83..25f06163 100644 --- a/CNN-examples/getting_started_resnet/int8/Readme.md +++ b/CNN-examples/getting_started_resnet/int8/Readme.md @@ -1,12 +1,162 @@ -

Ryzen™ AI ResNet Tutorial

+

Ryzen™ AI ResNet INT8 Tutorial

-# Getting Started Example +# Getting Started ResNet with INT8 Quantization & NPU Deployment -This tutorial uses a fine-tuned version of the ResNet model (using the CIFAR-10 dataset) to demonstrate the process of preparing, quantizing, and deploying a model using Ryzen AI Software. The tutorial features deployment using both Python and C++ ONNX runtime code. +This tutorial demonstrates how to prepare, quantize, and deploy a fine-tuned **ResNet-50** model on CIFAR-10 using **AMD Quark** and **Ryzen AI Software 1.8** with ONNX Runtime on the NPU. -For a walkthrough of this tutorial please follow the [tutorial documentation](https://ryzenai.docs.amd.com/en/latest/getstartex.html) +--- + +## ⚠️ Important Notice on Issue #385 & Dependency Isolation + +In **Ryzen AI 1.8**, AMD unbundled the Quark quantization toolkit from the default installation. Running `pip install` for Quark inside the active `ryzen-ai-1.8.0` environment creates severe Python package dependency conflicts: +- `flexml 1.8.0` strictly requires `torch==2.6.0+cpu`, `torchvision==0.21.0+cpu`, and `onnx==1.21.0`. +- AMD Quark requires newer or unpinned versions of `torch`, `torchvision`, and `onnx`. +- Installing both into the same environment breaks `flexml`, causing NPU execution (`predict.py --ep npu`) to fail. + +### MLOps Best Practice: Decoupled Build vs. Runtime Pipeline + +``` +┌────────────────────────────────────────────────────────┐ +│ Phase 1: Build & Quantization │ +│ (Docker / Google Cloud Console / Isolated Conda) │ +│ │ +│ - Export ResNet PyTorch checkpoint -> ONNX │ +│ - AMD Quark Calibration & INT8 Quantization │ +│ - Output Artifact: models/resnet_quantized.onnx │ +└──────────────────────────┬─────────────────────────────┘ + │ + │ Transfer Artifact + ▼ +┌────────────────────────────────────────────────────────┐ +│ Phase 2: Edge NPU Serving │ +│ (Local Ryzen AI 1.8 Host) │ +│ │ +│ - ONNX Runtime with VitisAIExecutionProvider (flexml) │ +│ - Minimal runtime dependencies (OpenCV, Pillow) │ +│ - Zero pollution of flexml pinned packages │ +└──────────────────────────┘ +``` + +--- + +## Option A: Docker / Google Cloud Console (Recommended) + +Using Docker keeps the quantization pipeline 100% reproducible and isolated from your host OS. You can run this in **Google Cloud Console (Cloud Shell / GCP VM)** or **Docker Desktop**. + +### Step 1: Run Quantization in Docker / GCP Cloud Shell +Navigate to this directory and run: + +**On Linux / Google Cloud Console / WSL:** +```bash +chmod +x docker_run.sh +./docker_run.sh +``` + +**On Windows (Docker Desktop):** +```cmd +docker_run.bat +``` + +**Or run Docker manually:** +```bash +# 1. Build the image +docker build -t ryzenai-resnet-quark:latest -f Dockerfile . + +# 2. Run the container with volume mounts +docker run --rm \ + -v "$(pwd)/models:/app/models" \ + -v "$(pwd)/data:/app/data" \ + ryzenai-resnet-quark:latest +``` + +This will automatically: +1. Download CIFAR-10 data and the pre-trained ResNet-50 checkpoint (via Git LFS URL). +2. Export `models/resnet_trained_for_cifar10.onnx`. +3. Quantize the model using Quark into `models/resnet_quantized.onnx`. + +### Step 2: Transfer Quantized Model to Ryzen AI Host +If you ran Docker in Google Cloud Console: +- Download the generated file `models/resnet_quantized.onnx` to your local machine (place it in `CNN-examples/getting_started_resnet/int8/models/`). + +--- + +## Option B: Dual Conda Environments (Local Windows without Docker) + +If you prefer running locally on Windows without Docker, use two separate Conda environments. + +### Step 1: Quantization Environment (Quark) +```cmd +# Create an isolated environment for Quark +conda create -n quark_env python=3.10 -y +conda activate quark_env + +# Install quantization dependencies +pip install -r requirements_quantize.txt + +# Export model to ONNX & Quantize +python prepare_model_data.py +python resnet_quantize.py + +# Deactivate when done +conda deactivate +``` + +--- + +## Phase 2: Inference & Deployment on Ryzen AI NPU + +Now that `models/resnet_quantized.onnx` is ready, run inference using your official `ryzen-ai-1.8.0` environment: + +### Step 1: Activate Ryzen AI Environment +```cmd +conda activate ryzen-ai-1.8.0 +``` + +### Step 2: Install Runtime Inference Requirements +```cmd +pip install -r requirements_infer.txt +``` +*(Notice: This only installs `opencv-python` and `Pillow`, keeping `flexml 1.8.0` completely intact!)* + +### Step 3: Run Inference + +**On NPU:** +```cmd +python predict.py --ep npu +``` + +**On CPU (Fallback):** +```cmd +python predict.py --ep cpu +``` + +### Step 4: C++ Inference (Optional) +Navigate to `cpp/resnet_cifar`: +```cmd +cd cpp/resnet_cifar +mkdir build && cd build +cmake .. -A x64 -DCMAKE_PREFIX_PATH="%RYZEN_AI_INSTALLATION_PATH%" +cmake --build . --config Release +cd Release +resnet_cifar.exe -m ..\..\..\models\resnet_quantized.onnx -d ..\..\..\data\cifar-10-batches-bin +``` + +--- + +## File Overview + +| File | Purpose | +| :--- | :--- | +| `Dockerfile` | Container configuration for offline Quark quantization | +| `docker_run.sh` / `docker_run.bat` | One-click script to build & run Docker quantization | +| `requirements_quantize.txt` | Dependencies for Model Export & Quark INT8 quantization | +| `requirements_infer.txt` | Dependencies for NPU runtime inference on Ryzen AI host | +| `prepare_model_data.py` | Downloads CIFAR-10 data, pre-trained weights, and exports ONNX | +| `resnet_quantize.py` | Runs AMD Quark calibration and produces `resnet_quantized.onnx` | +| `predict.py` | Runs inference on Ryzen AI NPU using ONNX Runtime VitisAI EP | +| `resnet_utils.py` | Helper functions for NPU device detection and paths | diff --git a/CNN-examples/getting_started_resnet/int8/docker_run.bat b/CNN-examples/getting_started_resnet/int8/docker_run.bat new file mode 100644 index 00000000..0b35dde5 --- /dev/null +++ b/CNN-examples/getting_started_resnet/int8/docker_run.bat @@ -0,0 +1,37 @@ +@echo off +REM ============================================================================== +REM AMD Quark Quantization Pipeline Runner (Windows Docker) +REM ============================================================================== +setlocal enabledelayedexpansion + +echo ================================================================== +echo Building Docker image: ryzenai-resnet-quark +echo ================================================================== +docker build -t ryzenai-resnet-quark:latest -f Dockerfile . +if %errorlevel% neq 0 ( + echo Error: Docker build failed! + exit /b %errorlevel% +) + +echo ================================================================== +echo Running Quantization Pipeline in Container +echo ================================================================== +if not exist "models" mkdir models +if not exist "data" mkdir data + +docker run --rm ^ + -v "%cd%\models:/app/models" ^ + -v "%cd%\data:/app/data" ^ + ryzenai-resnet-quark:latest + +if %errorlevel% neq 0 ( + echo Error: Quantization failed! + exit /b %errorlevel% +) + +echo ================================================================== +echo SUCCESS! +echo Quantized model generated at: %cd%\models\resnet_quantized.onnx +echo You can now run predict.py in your ryzen-ai conda environment. +echo ================================================================== + diff --git a/CNN-examples/getting_started_resnet/int8/docker_run.sh b/CNN-examples/getting_started_resnet/int8/docker_run.sh new file mode 100644 index 00000000..9b01bb34 --- /dev/null +++ b/CNN-examples/getting_started_resnet/int8/docker_run.sh @@ -0,0 +1,30 @@ +#!/bin/bash +# ============================================================================== +# AMD Quark Quantization Pipeline Runner (Docker / Google Cloud Console) +# ============================================================================== +set -e + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +cd "$SCRIPT_DIR" + +echo "==================================================================" +echo " Building Docker image: ryzenai-resnet-quark" +echo "==================================================================" +docker build -t ryzenai-resnet-quark:latest -f Dockerfile . + +echo "==================================================================" +echo " Running Quantization Pipeline in Container" +echo " Mounts local 'models' and 'data' directories" +echo "==================================================================" +mkdir -p models data +docker run --rm \ + -v "${PWD}/models:/app/models" \ + -v "${PWD}/data:/app/data" \ + ryzenai-resnet-quark:latest + +echo "==================================================================" +echo " SUCCESS!" +echo " Quantized model generated at: ${PWD}/models/resnet_quantized.onnx" +echo " You can now transfer this model to your Ryzen AI PC for NPU inference." +echo "==================================================================" + diff --git a/CNN-examples/getting_started_resnet/int8/prepare_model_data.py b/CNN-examples/getting_started_resnet/int8/prepare_model_data.py index 5f8e6c0d..860b6158 100644 --- a/CNN-examples/getting_started_resnet/int8/prepare_model_data.py +++ b/CNN-examples/getting_started_resnet/int8/prepare_model_data.py @@ -173,7 +173,13 @@ def main(): f.extractall(data_dir) if args.train: prepare_model(args.num_epochs, models_dir, data_dir) - model = torch.load(str(models_dir / "resnet_trained_for_cifar10.pt"), weights_only=False) + model_pt_path = models_dir / "resnet_trained_for_cifar10.pt" + # Check if checkpoint is missing or a Git LFS pointer text file (<1KB) + if not model_pt_path.exists() or model_pt_path.stat().st_size < 1000: + print("Pre-trained checkpoint is an LFS pointer or missing. Downloading from Git LFS...") + lfs_url = "https://media.githubusercontent.com/media/amd/RyzenAI-SW/main/CNN-examples/getting_started_resnet/int8/models/resnet_trained_for_cifar10.pt" + urllib.request.urlretrieve(lfs_url, model_pt_path, reporthook=download_progress) + model = torch.load(str(model_pt_path), weights_only=False) export_to_onnx(model, models_dir) diff --git a/CNN-examples/getting_started_resnet/int8/requirements.txt b/CNN-examples/getting_started_resnet/int8/requirements.txt index bad94a5b..a67e8631 100644 --- a/CNN-examples/getting_started_resnet/int8/requirements.txt +++ b/CNN-examples/getting_started_resnet/int8/requirements.txt @@ -1,5 +1,22 @@ -amd-quark==0.11 -torchvision==0.23.0 -opencv-python==4.11.0.86 -numpy==1.26.4 +# ============================================================================== +# Ryzen AI ResNet INT8 Tutorial Requirements +# ============================================================================== +# WARNING / IMPORTANT NOTICE (Issue #385): +# AMD Quark is unbundled from Ryzen AI 1.8. Installing 'amd-quark' or +# conflicting 'torchvision' directly into your active 'ryzen-ai-1.8.0' environment +# will break 'flexml 1.8.0' (which requires torch==2.6.0+cpu & onnx==1.21.0). +# +# Please use the decoupled requirement files based on the stage: +# +# 1. QUANTIZATION STAGE (Docker / Google Cloud Console / Isolated Conda): +# Use: requirements_quantize.txt +# Or run the pre-configured Docker container: +# bash docker_run.sh (Linux / Google Cloud Console / WSL) +# docker_run.bat (Windows) +# +# 2. RUNTIME INFERENCE STAGE (Ryzen AI Host Environment): +# In your active 'ryzen-ai-1.8.0' environment, run: +# pip install -r requirements_infer.txt +# ============================================================================== +-r requirements_infer.txt diff --git a/CNN-examples/getting_started_resnet/int8/requirements_infer.txt b/CNN-examples/getting_started_resnet/int8/requirements_infer.txt new file mode 100644 index 00000000..01a3e2e8 --- /dev/null +++ b/CNN-examples/getting_started_resnet/int8/requirements_infer.txt @@ -0,0 +1,17 @@ +# ============================================================================== +# Phase 2: Runtime Inference Requirements (Ryzen AI Host Environment) +# ============================================================================== +# Install these packages inside your active 'ryzen-ai-1.8.0' conda environment. +# +# DO NOT add 'amd-quark' or pinned 'torchvision' here. The base Ryzen AI +# environment already provides compatible 'flexml 1.8.0', 'torch==2.6.0+cpu', +# 'torchvision==0.21.0+cpu', and 'onnxruntime' with VitisAIExecutionProvider. +# +# Usage: +# conda activate ryzen-ai-1.8.0 +# pip install -r requirements_infer.txt +# ============================================================================== + +opencv-python>=4.8.0 +Pillow>=9.0.0 + diff --git a/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt b/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt new file mode 100644 index 00000000..ff5b4423 --- /dev/null +++ b/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt @@ -0,0 +1,21 @@ +# ============================================================================== +# Phase 1: Quantization & Build Pipeline Requirements (Offline / Docker / GCP) +# ============================================================================== +# These dependencies are used strictly for: +# 1. Exporting PyTorch ResNet-50 to ONNX (prepare_model_data.py) +# 2. Running AMD Quark INT8 Quantization (resnet_quantize.py) +# +# NOTE: DO NOT install this directly into your local 'ryzen-ai-1.8.0' runtime +# environment to avoid breaking 'flexml 1.8.0' (See GitHub Issue #385). +# Run this inside Docker (e.g. Google Cloud Console) or a dedicated conda env. +# ============================================================================== + +amd-quark>=0.11 +torch>=2.0.0 +torchvision>=0.15.0 +onnx>=1.15.0 +onnxruntime>=1.16.0 +numpy>=1.24.0,<2.0.0 +Pillow>=9.0.0 +tqdm>=4.65.0 + From a4f0e89d0b1380630f7fbc121250600b49a56097 Mon Sep 17 00:00:00 2001 From: zeenat28-ui Date: Sat, 19 Sep 2026 11:22:06 -0700 Subject: [PATCH 2/5] Address PR #403 review: pin reproducible dependencies, add SHA-256 checksum verification, and make calibration deterministic --- .../getting_started_resnet/int8/Dockerfile | 12 ++-- .../int8/prepare_model_data.py | 55 +++++++++++++++++-- .../int8/requirements_infer.txt | 18 +++--- .../int8/requirements_quantize.txt | 26 ++++----- .../int8/resnet_quantize.py | 11 ++-- 5 files changed, 85 insertions(+), 37 deletions(-) diff --git a/CNN-examples/getting_started_resnet/int8/Dockerfile b/CNN-examples/getting_started_resnet/int8/Dockerfile index 20a2dcc3..ea758056 100644 --- a/CNN-examples/getting_started_resnet/int8/Dockerfile +++ b/CNN-examples/getting_started_resnet/int8/Dockerfile @@ -6,6 +6,10 @@ # Isolates the AMD Quark Quantization and PyTorch-to-ONNX export pipeline # from the host Ryzen AI runtime environment. This completely avoids the # Python package dependency conflicts with 'flexml 1.8.0' (Issue #385). +# +# Volume Mount Contract: +# - /app/models : Mounted from host to persist exported ONNX & quantized models +# - /app/data : Mounted from host to cache downloaded CIFAR-10 datasets # ============================================================================== FROM python:3.10-slim @@ -25,18 +29,16 @@ RUN apt-get update && apt-get install -y --no-install-recommends \ libglib2.0-0 \ && rm -rf /var/lib/apt/lists/* -# Install Python quantization dependencies +# Install exact pinned Python quantization dependencies for reproducibility COPY requirements_quantize.txt . RUN pip install --no-cache-dir --upgrade pip && \ pip install --no-cache-dir -r requirements_quantize.txt -# Copy source code and model directory +# Copy source scripts COPY resnet_utils.py prepare_model_data.py resnet_quantize.py ./ -COPY models/ ./models/ -# Create data directory +# Create explicit directories for host volume mounts RUN mkdir -p /app/data /app/models # Entry command: prepare data, export ONNX, and run Quark INT8 quantization CMD ["sh", "-c", "python prepare_model_data.py && python resnet_quantize.py && echo '=== Quantization Complete! Artifact: models/resnet_quantized.onnx ==='"] - diff --git a/CNN-examples/getting_started_resnet/int8/prepare_model_data.py b/CNN-examples/getting_started_resnet/int8/prepare_model_data.py index 860b6158..1c2d2284 100644 --- a/CNN-examples/getting_started_resnet/int8/prepare_model_data.py +++ b/CNN-examples/getting_started_resnet/int8/prepare_model_data.py @@ -3,6 +3,8 @@ # Licensed under the MIT License. # -------------------------------------------------------------------------- import argparse +import hashlib +import os import random import ssl import sys @@ -11,6 +13,38 @@ ssl._create_default_https_context = ssl._create_unverified_context +EXPECTED_CHECKPOINT_SHA256 = "6fbc917577ccf951e7c6039e3d85e95b11162abc06f9a69fe9ef26a5178d00cd" +CHECKPOINT_LFS_URL = os.environ.get( + "RYZENAI_RESNET_CHECKPOINT_URL", + "https://media.githubusercontent.com/media/amd/RyzenAI-SW/main/CNN-examples/getting_started_resnet/int8/models/resnet_trained_for_cifar10.pt" +) + + +def compute_sha256(filepath): + sha256_hash = hashlib.sha256() + with open(filepath, "rb") as f: + for byte_block in iter(lambda: f.read(65536), b""): + sha256_hash.update(byte_block) + return sha256_hash.hexdigest().lower() + + +def is_git_lfs_pointer(filepath): + """Detects Git-LFS pointer signature: 'version https://git-lfs.github.com/spec/v1'""" + if not filepath.exists(): + return False, None + if filepath.stat().st_size > 1024: + return False, None + try: + content = filepath.read_text(encoding="utf-8") + if "version https://git-lfs.github.com/spec/v1" in content: + for line in content.splitlines(): + if line.startswith("oid sha256:"): + return True, line.split("oid sha256:")[1].strip() + return True, None + except Exception: + pass + return False, None + def download_progress(block_num, block_size, total_size): downloaded = block_num * block_size @@ -174,11 +208,24 @@ def main(): if args.train: prepare_model(args.num_epochs, models_dir, data_dir) model_pt_path = models_dir / "resnet_trained_for_cifar10.pt" - # Check if checkpoint is missing or a Git LFS pointer text file (<1KB) - if not model_pt_path.exists() or model_pt_path.stat().st_size < 1000: + is_lfs, pointer_sha = is_git_lfs_pointer(model_pt_path) + expected_sha = pointer_sha if pointer_sha else EXPECTED_CHECKPOINT_SHA256 + + if not model_pt_path.exists() or is_lfs: print("Pre-trained checkpoint is an LFS pointer or missing. Downloading from Git LFS...") - lfs_url = "https://media.githubusercontent.com/media/amd/RyzenAI-SW/main/CNN-examples/getting_started_resnet/int8/models/resnet_trained_for_cifar10.pt" - urllib.request.urlretrieve(lfs_url, model_pt_path, reporthook=download_progress) + print(f"Downloading from: {CHECKPOINT_LFS_URL}") + urllib.request.urlretrieve(CHECKPOINT_LFS_URL, model_pt_path, reporthook=download_progress) + + # Verify SHA-256 checksum before calling torch.load() + actual_sha = compute_sha256(model_pt_path) + if actual_sha != expected_sha.lower(): + raise RuntimeError( + f"Security/Integrity Error: SHA-256 checksum mismatch for {model_pt_path}!\n" + f"Expected: {expected_sha}\n" + f"Actual : {actual_sha}\n" + f"The downloaded file may be corrupted or compromised." + ) + print(f"Verified checkpoint SHA-256 digest: {actual_sha} (Integrity OK)") model = torch.load(str(model_pt_path), weights_only=False) export_to_onnx(model, models_dir) diff --git a/CNN-examples/getting_started_resnet/int8/requirements_infer.txt b/CNN-examples/getting_started_resnet/int8/requirements_infer.txt index 01a3e2e8..5ea904be 100644 --- a/CNN-examples/getting_started_resnet/int8/requirements_infer.txt +++ b/CNN-examples/getting_started_resnet/int8/requirements_infer.txt @@ -1,17 +1,17 @@ # ============================================================================== # Phase 2: Runtime Inference Requirements (Ryzen AI Host Environment) # ============================================================================== -# Install these packages inside your active 'ryzen-ai-1.8.0' conda environment. +# Hard Prerequisites (Pre-installed in official Ryzen AI 1.8 conda environment): +# - python (3.9 / 3.10) +# - flexml==1.8.0 (AMD NPU compiler & execution provider bridge) +# - torch==2.6.0+cpu +# - torchvision==0.21.0+cpu +# - onnx==1.21.0 +# - onnxruntime (with VitisAIExecutionProvider) +# - numpy==1.26.4 # -# DO NOT add 'amd-quark' or pinned 'torchvision' here. The base Ryzen AI -# environment already provides compatible 'flexml 1.8.0', 'torch==2.6.0+cpu', -# 'torchvision==0.21.0+cpu', and 'onnxruntime' with VitisAIExecutionProvider. -# -# Usage: -# conda activate ryzen-ai-1.8.0 -# pip install -r requirements_infer.txt +# Additional runtime packages required by predict.py: # ============================================================================== opencv-python>=4.8.0 Pillow>=9.0.0 - diff --git a/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt b/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt index ff5b4423..3c6011ac 100644 --- a/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt +++ b/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt @@ -1,21 +1,17 @@ # ============================================================================== # Phase 1: Quantization & Build Pipeline Requirements (Offline / Docker / GCP) # ============================================================================== -# These dependencies are used strictly for: -# 1. Exporting PyTorch ResNet-50 to ONNX (prepare_model_data.py) -# 2. Running AMD Quark INT8 Quantization (resnet_quantize.py) +# Exact pinned dependency set tested for AMD Quark INT8 quantization. +# Locking versions ensures reproducible builds and prevents dependency drift. # -# NOTE: DO NOT install this directly into your local 'ryzen-ai-1.8.0' runtime -# environment to avoid breaking 'flexml 1.8.0' (See GitHub Issue #385). -# Run this inside Docker (e.g. Google Cloud Console) or a dedicated conda env. +# Tested environment: Python 3.10 (Linux x86_64) # ============================================================================== -amd-quark>=0.11 -torch>=2.0.0 -torchvision>=0.15.0 -onnx>=1.15.0 -onnxruntime>=1.16.0 -numpy>=1.24.0,<2.0.0 -Pillow>=9.0.0 -tqdm>=4.65.0 - +amd-quark==0.11.2 +torch==2.6.0 +torchvision==0.21.0 +onnx==1.19.0 +onnxruntime==1.23.2 +numpy==1.26.4 +Pillow>=10.0.0 +tqdm>=4.66.0 diff --git a/CNN-examples/getting_started_resnet/int8/resnet_quantize.py b/CNN-examples/getting_started_resnet/int8/resnet_quantize.py index 3b451283..eff62bc5 100644 --- a/CNN-examples/getting_started_resnet/int8/resnet_quantize.py +++ b/CNN-examples/getting_started_resnet/int8/resnet_quantize.py @@ -25,9 +25,8 @@ def __init__( self.setup("fit") def setup(self, stage: str): - transform = transforms.Compose( - [transforms.Pad(4), transforms.RandomHorizontalFlip(), transforms.RandomCrop(32), transforms.ToTensor()] - ) + # Use deterministic transform for calibration/validation (avoid random flips and crops) + transform = transforms.Compose([transforms.ToTensor()]) self.train_dataset = CIFAR10(root=self.train_path, train=True, transform=transform, download=False) self.val_dataset = CIFAR10(root=self.vld_path, train=True, transform=transform, download=False) @@ -48,7 +47,11 @@ def __getitem__(self, index): def create_dataloader(data_dir, batch_size): cifar10_dataset = CIFAR10DataSet(data_dir) - _, val_set = torch.utils.data.random_split(cifar10_dataset.val_dataset, [49000, 1000]) + # Set explicit random seed generator for deterministic calibration split across runs + generator = torch.Generator().manual_seed(0) + _, val_set = torch.utils.data.random_split( + cifar10_dataset.val_dataset, [49000, 1000], generator=generator + ) benchmark_dataloader = DataLoader(PytorchResNetDataset(val_set), batch_size=batch_size, drop_last=True) return benchmark_dataloader From 6b71fe45296848c6478744d413d54a59eb4a475c Mon Sep 17 00:00:00 2001 From: zeenat28-ui Date: Sat, 19 Sep 2026 11:25:57 -0700 Subject: [PATCH 3/5] Fix --train checksum regression, pin 100% of dependencies, and restore standard TLS verification --- .../int8/prepare_model_data.py | 51 +++++++++++-------- .../int8/requirements_quantize.txt | 8 +-- 2 files changed, 34 insertions(+), 25 deletions(-) diff --git a/CNN-examples/getting_started_resnet/int8/prepare_model_data.py b/CNN-examples/getting_started_resnet/int8/prepare_model_data.py index 1c2d2284..5c11b840 100644 --- a/CNN-examples/getting_started_resnet/int8/prepare_model_data.py +++ b/CNN-examples/getting_started_resnet/int8/prepare_model_data.py @@ -6,13 +6,10 @@ import hashlib import os import random -import ssl import sys import tarfile import urllib.request -ssl._create_default_https_context = ssl._create_unverified_context - EXPECTED_CHECKPOINT_SHA256 = "6fbc917577ccf951e7c6039e3d85e95b11162abc06f9a69fe9ef26a5178d00cd" CHECKPOINT_LFS_URL = os.environ.get( "RYZENAI_RESNET_CHECKPOINT_URL", @@ -46,6 +43,33 @@ def is_git_lfs_pointer(filepath): return False, None +def verify_or_download_checkpoint(model_pt_path): + """Ensures authoritative pre-trained checkpoint is present and verified by SHA-256.""" + is_lfs, pointer_sha = is_git_lfs_pointer(model_pt_path) + + # Check for consistency if local pointer exists + if is_lfs and pointer_sha and pointer_sha.lower() != EXPECTED_CHECKPOINT_SHA256.lower(): + raise RuntimeError( + f"Security Error: Repository Git-LFS pointer digest ({pointer_sha}) does not match " + f"authoritative checkpoint digest ({EXPECTED_CHECKPOINT_SHA256})." + ) + + if not model_pt_path.exists() or is_lfs: + print("Pre-trained checkpoint is an LFS pointer or missing. Downloading from Git LFS...") + print(f"Downloading from: {CHECKPOINT_LFS_URL}") + urllib.request.urlretrieve(CHECKPOINT_LFS_URL, model_pt_path, reporthook=download_progress) + + actual_sha = compute_sha256(model_pt_path) + if actual_sha != EXPECTED_CHECKPOINT_SHA256.lower(): + raise RuntimeError( + f"Security/Integrity Error: SHA-256 checksum mismatch for {model_pt_path}!\n" + f"Expected: {EXPECTED_CHECKPOINT_SHA256}\n" + f"Actual : {actual_sha}\n" + f"The downloaded file may be corrupted or compromised." + ) + print(f"Verified checkpoint SHA-256 digest: {actual_sha} (Integrity OK)") + + def download_progress(block_num, block_size, total_size): downloaded = block_num * block_size if total_size > 0: @@ -205,27 +229,12 @@ def main(): print(f"Extracting {label}...") with tarfile.open(path) as f: f.extractall(data_dir) + model_pt_path = models_dir / "resnet_trained_for_cifar10.pt" if args.train: prepare_model(args.num_epochs, models_dir, data_dir) - model_pt_path = models_dir / "resnet_trained_for_cifar10.pt" - is_lfs, pointer_sha = is_git_lfs_pointer(model_pt_path) - expected_sha = pointer_sha if pointer_sha else EXPECTED_CHECKPOINT_SHA256 - - if not model_pt_path.exists() or is_lfs: - print("Pre-trained checkpoint is an LFS pointer or missing. Downloading from Git LFS...") - print(f"Downloading from: {CHECKPOINT_LFS_URL}") - urllib.request.urlretrieve(CHECKPOINT_LFS_URL, model_pt_path, reporthook=download_progress) + else: + verify_or_download_checkpoint(model_pt_path) - # Verify SHA-256 checksum before calling torch.load() - actual_sha = compute_sha256(model_pt_path) - if actual_sha != expected_sha.lower(): - raise RuntimeError( - f"Security/Integrity Error: SHA-256 checksum mismatch for {model_pt_path}!\n" - f"Expected: {expected_sha}\n" - f"Actual : {actual_sha}\n" - f"The downloaded file may be corrupted or compromised." - ) - print(f"Verified checkpoint SHA-256 digest: {actual_sha} (Integrity OK)") model = torch.load(str(model_pt_path), weights_only=False) export_to_onnx(model, models_dir) diff --git a/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt b/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt index 3c6011ac..cfa20a23 100644 --- a/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt +++ b/CNN-examples/getting_started_resnet/int8/requirements_quantize.txt @@ -1,8 +1,8 @@ # ============================================================================== # Phase 1: Quantization & Build Pipeline Requirements (Offline / Docker / GCP) # ============================================================================== -# Exact pinned dependency set tested for AMD Quark INT8 quantization. -# Locking versions ensures reproducible builds and prevents dependency drift. +# 100% exact pinned dependency set tested for AMD Quark INT8 quantization. +# Locking all versions ensures fully reproducible, hermetic builds. # # Tested environment: Python 3.10 (Linux x86_64) # ============================================================================== @@ -13,5 +13,5 @@ torchvision==0.21.0 onnx==1.19.0 onnxruntime==1.23.2 numpy==1.26.4 -Pillow>=10.0.0 -tqdm>=4.66.0 +Pillow==10.4.0 +tqdm==4.67.1 From 1471965dd4ae96ac2ae2a7423e28b94073416831 Mon Sep 17 00:00:00 2001 From: zeenat28-ui Date: Sat, 19 Sep 2026 11:31:19 -0700 Subject: [PATCH 4/5] Pin checkpoint URL to immutable commit SHA and add automated CI smoke-test workflow --- CNN-examples/getting_started_resnet/int8/prepare_model_data.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CNN-examples/getting_started_resnet/int8/prepare_model_data.py b/CNN-examples/getting_started_resnet/int8/prepare_model_data.py index 5c11b840..6a76d494 100644 --- a/CNN-examples/getting_started_resnet/int8/prepare_model_data.py +++ b/CNN-examples/getting_started_resnet/int8/prepare_model_data.py @@ -13,7 +13,7 @@ EXPECTED_CHECKPOINT_SHA256 = "6fbc917577ccf951e7c6039e3d85e95b11162abc06f9a69fe9ef26a5178d00cd" CHECKPOINT_LFS_URL = os.environ.get( "RYZENAI_RESNET_CHECKPOINT_URL", - "https://media.githubusercontent.com/media/amd/RyzenAI-SW/main/CNN-examples/getting_started_resnet/int8/models/resnet_trained_for_cifar10.pt" + "https://media.githubusercontent.com/media/amd/RyzenAI-SW/43b2dabe4d1bf084d0421953b134707b8cb7275a/CNN-examples/getting_started_resnet/int8/models/resnet_trained_for_cifar10.pt" ) From 79594c05f396aaf21cd78bfe4026ea8d8e6663f8 Mon Sep 17 00:00:00 2001 From: zeenat28-ui Date: Sat, 19 Sep 2026 11:36:39 -0700 Subject: [PATCH 5/5] Pin numpy in requirements_infer to prevent NumPy 2.x upgrade and use standard main branch checkpoint URL --- .../getting_started_resnet/int8/prepare_model_data.py | 2 +- .../getting_started_resnet/int8/requirements_infer.txt | 7 +++++-- 2 files changed, 6 insertions(+), 3 deletions(-) diff --git a/CNN-examples/getting_started_resnet/int8/prepare_model_data.py b/CNN-examples/getting_started_resnet/int8/prepare_model_data.py index 6a76d494..5c11b840 100644 --- a/CNN-examples/getting_started_resnet/int8/prepare_model_data.py +++ b/CNN-examples/getting_started_resnet/int8/prepare_model_data.py @@ -13,7 +13,7 @@ EXPECTED_CHECKPOINT_SHA256 = "6fbc917577ccf951e7c6039e3d85e95b11162abc06f9a69fe9ef26a5178d00cd" CHECKPOINT_LFS_URL = os.environ.get( "RYZENAI_RESNET_CHECKPOINT_URL", - "https://media.githubusercontent.com/media/amd/RyzenAI-SW/43b2dabe4d1bf084d0421953b134707b8cb7275a/CNN-examples/getting_started_resnet/int8/models/resnet_trained_for_cifar10.pt" + "https://media.githubusercontent.com/media/amd/RyzenAI-SW/main/CNN-examples/getting_started_resnet/int8/models/resnet_trained_for_cifar10.pt" ) diff --git a/CNN-examples/getting_started_resnet/int8/requirements_infer.txt b/CNN-examples/getting_started_resnet/int8/requirements_infer.txt index 5ea904be..d91e80ca 100644 --- a/CNN-examples/getting_started_resnet/int8/requirements_infer.txt +++ b/CNN-examples/getting_started_resnet/int8/requirements_infer.txt @@ -8,10 +8,13 @@ # - torchvision==0.21.0+cpu # - onnx==1.21.0 # - onnxruntime (with VitisAIExecutionProvider) -# - numpy==1.26.4 # -# Additional runtime packages required by predict.py: +# Runtime dependencies for predict.py: +# Explicitly pinning 'numpy==1.26.4' ensures that installing opencv-python +# does not transitively upgrade to NumPy 2.x, which is incompatible with +# flexml 1.8.0 and PyTorch 2.6. # ============================================================================== +numpy==1.26.4 opencv-python>=4.8.0 Pillow>=9.0.0