From d4dd4735e9d7d308c00a04fa30f314b86f26a723 Mon Sep 17 00:00:00 2001 From: "Guillaume Meyer (The Opinionated Man)" <1385518+guillaumemeyer@users.noreply.github.com> Date: Mon, 17 Aug 2026 16:48:26 -0700 Subject: [PATCH] fix: Windows setup_ctrlregen.ps1 torch install (probe published indices, keep CUDA torch) (#124) Three independent failure modes from #117: - $ErrorActionPreference 'Stop' + 2>$null on a native command aborts the script on torch's harmless stderr warnings (e.g. "Failed to initialize NumPy" when torch is installed before numpy). Run the probes through a new Invoke-NativeQuiet helper that lowers EAP to 'Continue' for the block and restores it afterwards. - The wheel index tag was derived from the driver's CUDA version, e.g. cu131 for a 13.1 driver, which does not exist (HTTP 403) and silently fell back to the default index, i.e. the CPU build on Windows. Probe the published indices and pick the highest one <= driver that answers HTTP 200; cu126 is still forced below compute capability 7.5. - Installing torch alone let requirements-ctrlregen.txt resolve torchvision from PyPI, and torchvision pins an exact torch, so pip replaced the +cu build with a +cpu one while the script still exited 0. Install torch AND torchvision together from the chosen index, and verify after the requirements install that torch.cuda.is_available() is true - if a GPU was detected but torch ends up CPU-only, warn loudly and exit non-zero. Also add a CI step (windows-latest, pwsh) that parses the setup .ps1 scripts and asserts the post-install CUDA verification survives. Fixes #117 --- .github/workflows/ci.yml | 13 ++ README.md | 16 ++- service/scripts/setup_ctrlregen.ps1 | 205 +++++++++++++++++++--------- 3 files changed, 167 insertions(+), 67 deletions(-) diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index de60ae5..b26d793 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -41,6 +41,19 @@ jobs: python service/scripts/clean_file.py tests/fixtures/sample_ai.html -o "$env:TEMP/sample_ai.cleaned.html" && python service/scripts/clean_file.py tests/fixtures/sample_meta.svg -o "$env:TEMP/sample_meta.cleaned.svg" + - name: Validate setup .ps1 scripts + if: matrix.os == 'windows-latest' + shell: pwsh + run: | + $files = @('service/scripts/setup_ctrlregen.ps1', 'service/scripts/setup_synthid.ps1') + foreach ($f in $files) { + $tokens = $null; $errors = $null + [System.Management.Automation.Language.Parser]::ParseFile((Resolve-Path $f), [ref]$tokens, [ref]$errors) | Out-Null + if ($errors.Count) { $errors | ForEach-Object { Write-Error $_.Message }; exit 1 } + } + $ps1 = Get-Content service/scripts/setup_ctrlregen.ps1 -Raw + if ($ps1 -notmatch 'cuda\.is_available\(\)') { Write-Error 'post-install CUDA verification missing'; exit 1 } + - name: Audit dependencies (pip-audit) if: matrix.os == 'ubuntu-latest' run: | diff --git a/README.md b/README.md index 18a479e..e913f1b 100644 --- a/README.md +++ b/README.md @@ -330,12 +330,16 @@ NOAI_WATERMARK_DIR=~/noai-watermark \ On Windows use `setup_ctrlregen.ps1` (same flags as `-Dir`, `-Ref`, `-Python`); the venv lands in `.venv\Scripts\`, which `clean_image.py` already resolves. -It picks the torch wheel index from the GPU's **compute capability** rather -than the CUDA version `nvidia-smi` prints — that number is the maximum the -*driver* supports, and drivers are backward compatible, so deriving the wheel -tag from it installs `cu130` on a Pascal card whose kernels were dropped in -`cu128`. The script forces `cu126` below compute capability 7.5 and then -verifies the result with `torch.cuda.get_arch_list()`. +It probes the published PyTorch wheel indices and picks the highest one at or +below the CUDA version `nvidia-smi` prints that actually exists — that number +is the maximum the *driver* supports, and drivers are backward compatible, so a +driver reporting 13.1 (no published `cu131`) installs `cu130`. Below compute +capability 7.5 it forces `cu126`, the last index whose wheels still carry +Maxwell/Pascal/Volta kernels. It installs `torch` **and** `torchvision` +together from that index so the dependency install cannot swap them for CPU +builds from PyPI, then verifies after install that `torch.cuda.is_available()` +is true — if a GPU was detected but torch ends up CPU-only, the script warns +loudly and exits non-zero instead of pretending the setup succeeded. ### From `clean_image.py` diff --git a/service/scripts/setup_ctrlregen.ps1 b/service/scripts/setup_ctrlregen.ps1 index f0ffa6a..43f7b4a 100644 --- a/service/scripts/setup_ctrlregen.ps1 +++ b/service/scripts/setup_ctrlregen.ps1 @@ -1,28 +1,35 @@ <# .SYNOPSIS - Port a Windows de setup_ctrlregen.sh. + Windows port of setup_ctrlregen.sh. .DESCRIPTION - Bootstrap del checkout externo noai-watermark para el backend opcional de - eliminacion en dominio de pixel (CtrlRegen). + Bootstraps an external noai-watermark checkout for the optional CtrlRegen + pixel-domain removal backend. - El proyecto upstream (https://github.com/mertizci/noai-watermark) no incluye - fichero LICENSE, asi que su codigo se trata como all-rights-reserved y NO se - empaqueta en este repositorio. Este script lo clona en local e instala solo - las dependencias que necesita clean_ctrlregen.py. + The upstream project (https://github.com/mertizci/noai-watermark) does not + ship a LICENSE file, so its code is treated as all-rights-reserved and is NOT + bundled in this repository. This script clones it locally and installs only + the dependencies that clean_ctrlregen.py needs. - Diferencia con la version .sh: el venv de Windows vive en .venv\Scripts\ - en vez de .venv/bin/, y si el indice de torch para tu version de CUDA no - existe se cae al torch por defecto en vez de abortar. + Differences from the .sh version: the Windows venv lives in .venv\Scripts\ + instead of .venv/bin/. + + The torch index is not derived from the driver's CUDA version: the published + indices are probed (the highest one <= the driver's CUDA that answers HTTP + 200), and torch AND torchvision are installed together from that index so + installing requirements-ctrlregen.txt cannot replace them with CPU builds + from PyPI. If a GPU is detected but the final torch has no CUDA support, the + script warns loudly and exits with a non-zero exit code (never "Done" with a + silent CPU-only environment). .PARAMETER Dir - Directorio del checkout (por defecto: $env:NOAI_WATERMARK_DIR o ~\noai-watermark) + Checkout directory (default: $env:NOAI_WATERMARK_DIR or ~\noai-watermark) .PARAMETER Ref - Commit a usar (por defecto: el SHA fijado; no apuntes a una rama movil) + Commit to use (default: the pinned SHA; do not point at a moving branch) .PARAMETER Python - Interprete usado para crear el venv (por defecto: python) + Interpreter used to create the venv (default: python) #> [CmdletBinding()] param( @@ -37,7 +44,24 @@ $ScriptDir = Split-Path -Parent $MyInvocation.MyCommand.Path function Invoke-Checked { param([string]$What, [scriptblock]$Block) & $Block - if ($LASTEXITCODE -ne 0) { throw "$What fallo (codigo $LASTEXITCODE)" } + if ($LASTEXITCODE -ne 0) { throw "$What failed (exit code $LASTEXITCODE)" } +} + +# Runs a native command discarding its stderr safely. +# In Windows PowerShell, redirecting a native executable's stderr (2>$null, +# 2>&1, 2>file) wraps each line in an ErrorRecord; with +# $ErrorActionPreference = 'Stop' the first one aborts the script. Lower EAP to +# 'Continue' for just this block (e.g. torch warnings on stderr) and restore it +# afterwards. +function Invoke-NativeQuiet { + param([scriptblock]$Block) + $prev = $ErrorActionPreference + try { + $ErrorActionPreference = 'Continue' + & $Block 2>$null + } finally { + $ErrorActionPreference = $prev + } } if (-not $Dir) { @@ -50,89 +74,148 @@ if (-not (Test-Path $Dir)) { New-Item -ItemType Directory -Force -Path $Dir | Ou $Dir = (Resolve-Path $Dir).Path if (-not (Test-Path (Join-Path $Dir '.git'))) { - Write-Host "Clonando noai-watermark en $Dir (ref fijado: $Ref)" + Write-Host "Cloning noai-watermark into $Dir (pinned ref: $Ref)" Invoke-Checked 'git clone' { git clone --depth 1 --filter=blob:none --sparse https://github.com/mertizci/noai-watermark.git $Dir } Invoke-Checked 'git fetch' { git -C $Dir fetch --depth 1 origin $Ref } Invoke-Checked 'git checkout' { git -C $Dir checkout --detach $Ref } Invoke-Checked 'sparse-checkout' { git -C $Dir sparse-checkout set --no-cone '/src/' } $head = (git -C $Dir rev-parse HEAD).Trim() - if ($head -ne $Ref) { throw "error: se esperaba el ref fijado $Ref, se obtuvo $head" } + if ($head -ne $Ref) { throw "error: expected pinned ref $Ref, got $head" } } else { - Write-Host "Usando checkout existente: $Dir" + Write-Host "Using existing checkout: $Dir" } $venvPython = Join-Path $Dir '.venv\Scripts\python.exe' if (-not (Test-Path $venvPython)) { - Write-Host "Creando venv en $Dir\.venv" - Invoke-Checked 'crear venv' { & $Python -m venv (Join-Path $Dir '.venv') } + Write-Host "Creating venv at $Dir\.venv" + Invoke-Checked 'create venv' { & $Python -m venv (Join-Path $Dir '.venv') } } -Write-Host 'Instalando dependencias de Python' -# pip fijado (un --upgrade pip sin pin es un punto de deriva de cadena de suministro). +Write-Host 'Installing Python dependencies' +# Pinned pip (an unpinned --upgrade pip was a supply-chain drift point). Invoke-Checked 'pip install pip' { & $venvPython -m pip install --upgrade 'pip==26.2.1' } -# torch con el indice de plataforma correcto antes que el resto de pines. +# Install torch from the correct platform index before the remaining pins. # -# OJO: nvidia-smi reporta la CUDA MAXIMA QUE SOPORTA EL DRIVER, no la que hay -# que instalar; el driver es retrocompatible, asi que un wheel cu126 corre sin -# problema sobre un driver 13.0. Derivar el tag del wheel de esa cifra -como -# hace la version .sh- es incorrecto en general, y rompe de verdad en GPUs -# antiguas: los wheels cu128+ y CUDA 13 eliminaron Maxwell/Pascal/Volta, asi -# que en una Pascal (sm_61) instalarian un torch sin kernels para la tarjeta y -# fallaria en ejecucion con "no kernel image is available for execution". -# Por eso mandamos la compute capability, no la version del driver. +# NOTE: nvidia-smi reports the MAXIMUM CUDA the driver supports, not the CUDA +# to install; the driver is backward compatible, so a cu126 wheel runs fine on +# a 13.0 driver. Deriving the wheel tag from that number - as the .sh version +# does - is wrong in general: there is no index per driver version (e.g. a 13.1 +# driver would yield cu131, which does not exist, HTTP 403, and the script +# would fall back to the default torch, which is the CPU build on Windows). +# Instead the published indices are probed and the highest one <= the driver's +# CUDA is chosen. Also, cu128+ / CUDA 13 wheels dropped Maxwell/Pascal/Volta +# kernels, so on a Pascal (sm_61) they would install a torch without kernels +# for the card, failing at runtime with "no kernel image is available for +# execution". That is why the compute capability is used, not the driver +# version. +$knownTags = @( + @{ tag = 'cu130'; version = [version]'13.0' }, + @{ tag = 'cu129'; version = [version]'12.9' }, + @{ tag = 'cu128'; version = [version]'12.8' }, + @{ tag = 'cu126'; version = [version]'12.6' }, + @{ tag = 'cu124'; version = [version]'12.4' }, + @{ tag = 'cu121'; version = [version]'12.1' }, + @{ tag = 'cu118'; version = [version]'11.8' } +) + +# Picks the wheel index: the highest one <= the driver's CUDA answering 2xx. +# cc < 7.5 forces cu126 (last index with Maxwell/Pascal/Volta kernels). +function Select-TorchTag([double]$cc, [string]$driverCuda) { + if ($cc -and $cc -lt 7.5) { return 'cu126' } + if ($driverCuda -notmatch '^([0-9]+)\.([0-9]+)$') { return $null } + $driverVersion = [version]"$($Matches[1]).$($Matches[2])" + foreach ($c in $knownTags) { + if ($c.version -gt $driverVersion) { continue } + try { + $resp = Invoke-WebRequest -UseBasicParsing -Method Head -Uri "https://download.pytorch.org/whl/$($c.tag)" -TimeoutSec 15 -ErrorAction Stop + if ($resp.StatusCode -ge 200 -and $resp.StatusCode -lt 300) { return $c.tag } + } catch { } # index not published / network error: try the next one + } + return $null +} + $cuda = $null $cc = $null if (Get-Command nvidia-smi -ErrorAction SilentlyContinue) { $smi = (& nvidia-smi | Out-String) if ($smi -match 'CUDA Version:\s*([0-9]+\.[0-9]+)') { $cuda = $Matches[1] } - $capRaw = (& nvidia-smi --query-gpu=compute_cap --format=csv,noheader 2>$null | Select-Object -First 1) + $capRaw = (Invoke-NativeQuiet { & nvidia-smi --query-gpu=compute_cap --format=csv,noheader } | Select-Object -First 1) if ($capRaw -and ($capRaw.Trim() -match '^[0-9]+\.[0-9]+$')) { $cc = [double]$capRaw.Trim() } } +# Install torch AND torchvision together from the same index. If only torch +# were installed, the requirements install would resolve torchvision from PyPI, +# and torchvision pins an exact torch version, so pip would uninstall the +cu +# build and replace it with the +cpu one, silently and with exit 0. Installing +# them together closes that hole; the final verification below checks it too. $torchOk = $false if ($cuda) { if ($cc -and $cc -lt 7.5) { - # Ultimo indice cuyos wheels aun traen kernels de Maxwell/Pascal/Volta. $tag = 'cu126' - Write-Host "GPU NVIDIA compute capability $cc (anterior a Turing): forzando $tag," - Write-Host "porque los wheels cu128+ / CUDA 13 ya no incluyen kernels para esta tarjeta." + Write-Host "NVIDIA GPU with compute capability $cc (pre-Turing): forcing $tag," + Write-Host "because cu128+ / CUDA 13 wheels no longer ship kernels for this card." } else { - $tag = 'cu' + ($cuda -replace '\.', '') - Write-Host "GPU NVIDIA detectada (driver soporta CUDA $cuda); usando $tag" + $tag = Select-TorchTag $cc $cuda + if ($tag) { + Write-Host "NVIDIA GPU detected (driver supports CUDA $cuda); using published index $tag" + } else { + Write-Warning "no published torch index <= CUDA $cuda; falling back to the default torch (likely CPU)" + } + } + if ($tag) { + $index = "https://download.pytorch.org/whl/$tag" + & $venvPython -m pip install torch torchvision --index-url $index + if ($LASTEXITCODE -eq 0) { $torchOk = $true } + else { Write-Warning "the index $index failed; falling back to the default torch" } } - $index = "https://download.pytorch.org/whl/$tag" - & $venvPython -m pip install torch --index-url $index - if ($LASTEXITCODE -eq 0) { $torchOk = $true } - else { Write-Warning "el indice $index fallo; se cae al torch por defecto" } } else { - Write-Host 'Sin GPU NVIDIA detectada; instalando torch por defecto (CPU)' + Write-Host 'No NVIDIA GPU detected; installing the default torch (CPU)' } if (-not $torchOk) { - Invoke-Checked 'pip install torch' { & $venvPython -m pip install torch } -} - -# Comprobacion decisiva: que el torch instalado traiga kernels para ESTA GPU. -if ($cc) { - $smTarget = 'sm_' + ($cc -replace '\.', '') - $archs = (& $venvPython -c "import torch; print(' '.join(torch.cuda.get_arch_list()))" 2>$null) - if ($LASTEXITCODE -eq 0 -and $archs) { - if ($archs -match [regex]::Escape($smTarget)) { - Write-Host "OK: torch incluye $smTarget (archs: $archs)" - } else { - Write-Warning "el torch instalado NO incluye $smTarget - CtrlRegen fallara en GPU." - Write-Warning "archs disponibles: $archs" - Write-Warning "reinstala con un indice mas antiguo, p.ej.:" - Write-Warning " & '$venvPython' -m pip install --force-reinstall torch --index-url https://download.pytorch.org/whl/cu126" - } - } + Invoke-Checked 'pip install torch torchvision' { & $venvPython -m pip install torch torchvision } } Invoke-Checked 'pip install requirements' { & $venvPython -m pip install -r (Join-Path $ScriptDir 'requirements-ctrlregen.txt') } +# Final verification (AFTER requirements, the step that could have replaced the +# +cu torch with a +cpu one): if a GPU is present but the final torch has no +# CUDA support, warn loudly and exit non-zero instead of pretending success +# with a silent CPU-only environment. +if ($cc -or $cuda) { + $probe = Invoke-NativeQuiet { + & $venvPython -c "import torch; print(torch.cuda.is_available()); print(' '.join(torch.cuda.get_arch_list()))" + } + if ($LASTEXITCODE -eq 0 -and $probe) { + $lines = @($probe) + $cudaOk = ($lines[0].Trim() -eq 'True') + $archs = $lines[1] + if (-not $cudaOk) { + Write-Warning 'NVIDIA GPU detected but torch.cuda.is_available() = False.' + Write-Warning 'CtrlRegen would run on CPU (very slow) or fail on GPU.' + Write-Warning 'Reinstall torch and torchvision together from a published index, e.g.:' + Write-Warning " & '$venvPython' -m pip install --force-reinstall torch torchvision --index-url https://download.pytorch.org/whl/cu130" + exit 1 + } + if ($cc) { + $smTarget = 'sm_' + ($cc -replace '\.', '') + if ($archs -and ($archs -notmatch [regex]::Escape($smTarget))) { + Write-Warning "the installed torch does NOT include $smTarget - CtrlRegen will fail on GPU." + Write-Warning "available archs: $archs" + Write-Warning "reinstall from an older index, e.g.:" + Write-Warning " & '$venvPython' -m pip install --force-reinstall torch torchvision --index-url https://download.pytorch.org/whl/cu126" + } elseif ($archs) { + Write-Host "OK: torch includes $smTarget (archs: $archs)" + } + } + } else { + Write-Warning 'could not verify torch after installing requirements' + } +} + Write-Host '' -Write-Host 'Listo. Para quitar una marca:' +Write-Host 'Done. Remove a watermark with:' Write-Host '' -Write-Host " `$env:NOAI_WATERMARK_DIR = '$Dir'" -Write-Host " & '$venvPython' '$ScriptDir\clean_ctrlregen.py' IMAGEN -o SALIDA" +Write-Host " \`$env:NOAI_WATERMARK_DIR = '$Dir'" +Write-Host " & '$venvPython' '$ScriptDir\clean_ctrlregen.py' IMAGE -o OUTPUT" \ No newline at end of file