From fe932d955495177b149b57481d60cae350e4db02 Mon Sep 17 00:00:00 2001 From: MrScripty Date: Sat, 26 Sep 2026 19:53:05 -0700 Subject: [PATCH 1/7] Expose FLUX adapter in Torch installation preview The desktop flow always selected Core even though the managed installer supports FLUX.2. Surface manager-supported adapter choices and validate the selected preview token. Record the Torch 2.14 CUDA, managed startup, gateway, and RAM admission findings without claiming image generation. --- .../upstream-version-manager-progress.md | 38 +++++++++ .../components/TorchInstallPreview.test.tsx | 77 +++++++++++++++++++ .../src/components/TorchInstallPreview.tsx | 27 ++++++- 3 files changed, 138 insertions(+), 4 deletions(-) diff --git a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md index 32b7cd26..b4c9e58a 100644 --- a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md +++ b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md @@ -40,6 +40,44 @@ fresh Torch installation or Tuldok generation was performed. This collector measures pip package transfers; managed Python/bootstrap and other Pumas network producers remain uncovered. +## 2026-09-26 — Installed Torch 2.14 image-serving gate check + +The selected local `v2.14.0` is `2.14.0+cu132` on managed Python 3.14.7 with +`adapter: none`. Its environment has no Diffusers, Transformers, Accelerate, or +PEFT distribution. A real CUDA tensor on the host RTX 5090 returned `5`; the +managed Pumas `trial_torch_runtime` RPC passed startup, health, protocol 3, and +the sidecar's `image_generation` handshake, then the trial profile was stopped. +The public Pumas `GET /v1/models` returned an empty model list. A direct socket +startup of the exact installed sidecar also returned HTTP 200 from `/health`. + +The library has the FLUX.2 Klein checkpoint, Qwen3-8B component package, and +standalone VAE. A real `serve_model` RPC for the FLUX checkpoint returned +`insufficient_memory` before model load: Pumas measured 66,815,033,344 bytes +total RAM at 38.48% usage, below the existing 42 GiB available-RAM gate. The +gateway still advertised no image model afterward. The temporary RPC and +sidecar processes were stopped. These checks do **not** qualify v2.14 image +generation or Tuldok display/save: the installed Core recipe lacks a supported +image adapter, and this host currently fails FLUX admission. + +The desktop Torch install preview previously hardcoded `adapter: none` despite +the managed installer supporting a separate `flux2` choice. The preview now +offers that choice when reported by the manager, preserves the manager's +supported default, sends the selected adapter to the one-use preview, and +rejects a returned selection for a different adapter. Existing installed tags +remain immutable; enabling images for the current v2.14 tag requires a managed +replacement after an isolated v2.14+FLUX qualification and enough available +RAM. No installed runtime, model, or public release was replaced here. + +Verification for the preview change passed 23 focused frontend tests, TypeScript, +ESLint on the touched files, production frontend/Electron builds, Linux artifact +checks, and extracted AppImage/deb resource plus RPC-health smokes. Independent +read-only review found no blocking issue in its adapter authorization or stale +request handling. The rebuilt local, unpublished AppImage SHA-256 is +`f1d452c7bdb116e63634188c9756506c3bea66944fdee13a3ef15fb49890494b`; +the Debian package SHA-256 is +`b010a8e86cddd5c07cd19e28da56b0c09f9e5f62613af5d6b82bd1192fe574b3`. +Neither package was used for a v2.14+FLUX install or real Tuldok generation. + ## 2026-09-26 — Bounded live transfer and packaged code check The latest successful `install-v2.14.0-1790470772611.log` used cached wheels, diff --git a/frontend/src/components/TorchInstallPreview.test.tsx b/frontend/src/components/TorchInstallPreview.test.tsx index 8c084ee7..c81b8e91 100644 --- a/frontend/src/components/TorchInstallPreview.test.tsx +++ b/frontend/src/components/TorchInstallPreview.test.tsx @@ -75,6 +75,7 @@ describe('TorchInstallPreview', () => { expect(getSelection).not.toHaveBeenCalled(); expect(screen.queryByRole('button', { name: 'Check selected combination' })).not.toBeInTheDocument(); expect(screen.queryByRole('combobox')).not.toBeInTheDocument(); + expect(screen.queryByRole('radio', { name: 'FLUX.2 image generation' })).not.toBeInTheDocument(); expect(screen.queryByText(/exact artifacts resolved/i)).not.toBeInTheDocument(); }); @@ -93,6 +94,82 @@ describe('TorchInstallPreview', () => { expect(getSelection).toHaveBeenCalledWith({ tag: 'v2.14.0', build: 'auto', python: 'auto', adapter: 'none' }); }); + it('offers FLUX.2 only when supported and keeps the current manager Core default', async () => { + getOptions.mockResolvedValue({ ...options, adapters: ['none', 'flux2'] }); + const onInstall = vi.fn(); + render(); + + expect(await screen.findByRole('radio', { name: 'Core Torch' })).toBeChecked(); + expect(screen.getByRole('radio', { name: 'FLUX.2 image generation' })).not.toBeChecked(); + fireEvent.click(screen.getByRole('button', { name: 'Install Torch' })); + await waitFor(() => expect(onInstall).toHaveBeenCalledWith('selection-token')); + expect(getSelection).toHaveBeenCalledWith({ tag: 'v2.14.0', build: 'auto', python: 'auto', adapter: 'none' }); + }); + + it('honors a supported manager default adapter and falls back to Core if it is unsupported', async () => { + getOptions.mockResolvedValueOnce({ ...options, adapters: ['none', 'flux2'], defaultAdapter: 'flux2' }); + const onInstall = vi.fn(); + const { rerender } = render(); + + expect(await screen.findByRole('radio', { name: 'FLUX.2 image generation' })).toBeChecked(); + fireEvent.click(screen.getByRole('button', { name: 'Install Torch' })); + await waitFor(() => expect(onInstall).toHaveBeenCalledWith('selection-token')); + expect(getSelection).toHaveBeenLastCalledWith({ tag: 'v2.14.0', build: 'auto', python: 'auto', adapter: 'flux2' }); + + getOptions.mockResolvedValueOnce({ ...options, defaultAdapter: 'flux2' }); + rerender(); + await waitFor(() => expect(screen.getByRole('button', { name: 'Install Torch' })).toBeEnabled()); + expect(screen.queryByRole('radio', { name: 'FLUX.2 image generation' })).not.toBeInTheDocument(); + fireEvent.click(screen.getByRole('button', { name: 'Install Torch' })); + await waitFor(() => expect(onInstall).toHaveBeenCalledTimes(2)); + expect(getSelection).toHaveBeenLastCalledWith({ tag: 'v2.14.1', build: 'auto', python: 'auto', adapter: 'none' }); + }); + + it('previews the selected FLUX.2 adapter and rejects a token for another adapter', async () => { + getOptions.mockResolvedValue({ ...options, adapters: ['none', 'flux2'] }); + const onInstall = vi.fn(); + getSelection.mockResolvedValueOnce(readySelection({ tag: 'v2.14.0', build: 'auto', python: 'auto', adapter: 'none' })); + render(); + + fireEvent.click(await screen.findByRole('radio', { name: 'FLUX.2 image generation' })); + fireEvent.click(screen.getByRole('button', { name: 'Install Torch' })); + expect(getSelection).toHaveBeenCalledWith({ tag: 'v2.14.0', build: 'auto', python: 'auto', adapter: 'flux2' }); + expect(await screen.findByRole('alert')).toHaveTextContent('Torch selection could not be confirmed'); + expect(onInstall).not.toHaveBeenCalled(); + + fireEvent.click(screen.getByRole('button', { name: 'Install Torch' })); + await waitFor(() => expect(onInstall).toHaveBeenCalledWith('selection-token')); + }); + + it('resets the adapter for another tag and ignores its stale preview response', async () => { + getOptions.mockResolvedValue({ ...options, adapters: ['none', 'flux2'] }); + let resolveSelection!: (value: TorchRuntimePreviewOutcome) => void; + getSelection.mockImplementationOnce(() => new Promise((resolve) => { resolveSelection = resolve; })); + const onInstall = vi.fn(); + const { rerender } = render(); + + fireEvent.click(await screen.findByRole('radio', { name: 'FLUX.2 image generation' })); + fireEvent.click(screen.getByRole('button', { name: 'Install Torch' })); + rerender(); + expect(await screen.findByRole('radio', { name: 'Core Torch' })).toBeChecked(); + await act(async () => { resolveSelection(readySelection({ tag: 'v2.14.0', build: 'auto', python: 'auto', adapter: 'flux2' })); }); + expect(onInstall).not.toHaveBeenCalled(); + fireEvent.click(screen.getByRole('button', { name: 'Install Torch' })); + await waitFor(() => expect(onInstall).toHaveBeenCalledWith('selection-token')); + expect(getSelection).toHaveBeenLastCalledWith({ tag: 'v2.14.1', build: 'auto', python: 'auto', adapter: 'none' }); + }); + + it('ignores options returned for a previous tag', async () => { + let resolveOldOptions!: (value: TorchRuntimeOptions) => void; + getOptions.mockImplementationOnce(() => new Promise((resolve) => { resolveOldOptions = resolve; })); + const { rerender } = render(); + rerender(); + + expect(await screen.findByRole('button', { name: 'Install Torch' })).toBeEnabled(); + await act(async () => { resolveOldOptions({ ...options, adapters: ['none', 'flux2'] }); }); + expect(screen.queryByRole('radio', { name: 'FLUX.2 image generation' })).not.toBeInTheDocument(); + }); + it('shows immediate pending status and starts the install with the selection token', async () => { let resolveSelection!: (value: TorchRuntimePreviewOutcome) => void; getSelection.mockImplementationOnce(() => new Promise((resolve) => { resolveSelection = resolve; })); diff --git a/frontend/src/components/TorchInstallPreview.tsx b/frontend/src/components/TorchInstallPreview.tsx index 00766cd2..1e5ce875 100644 --- a/frontend/src/components/TorchInstallPreview.tsx +++ b/frontend/src/components/TorchInstallPreview.tsx @@ -24,6 +24,7 @@ function errorText(error: unknown): string { export function TorchInstallPreview({ tag, onBack, onInstall }: TorchInstallPreviewProps) { const [runtimeOptions, setRuntimeOptions] = useState(null); + const [adapter, setAdapter] = useState<'none' | 'flux2'>('none'); const [loadingOptions, setLoadingOptions] = useState(true); const [starting, setStarting] = useState(false); const [error, setError] = useState(null); @@ -34,12 +35,16 @@ export function TorchInstallPreview({ tag, onBack, onInstall }: TorchInstallPrev let active = true; requestNumber.current += 1; setRuntimeOptions(null); + setAdapter('none'); setLoadingOptions(true); setStarting(false); setError(null); void api.get_torch_runtime_options().then((options) => { - if (active) setRuntimeOptions(options); + if (active) { + setAdapter(options.adapters.includes(options.defaultAdapter) ? options.defaultAdapter : 'none'); + setRuntimeOptions(options); + } }).catch((cause: unknown) => { if (active) setError(`Torch runtime choices unavailable: ${errorText(cause)}`); }).finally(() => { @@ -53,9 +58,10 @@ export function TorchInstallPreview({ tag, onBack, onInstall }: TorchInstallPrev }, [tag, attempt]); const startInstallation = async () => { - if (!runtimeOptions || starting) return; + if (!runtimeOptions || starting || (adapter === 'flux2' && !runtimeOptions.adapters.includes('flux2'))) return; const currentRequest = ++requestNumber.current; const build = runtimeOptions.defaultBuild || 'auto'; + const selectedAdapter = adapter; setStarting(true); setError(null); @@ -64,7 +70,7 @@ export function TorchInstallPreview({ tag, onBack, onInstall }: TorchInstallPrev tag, build, python: 'auto', - adapter: 'none', + adapter: selectedAdapter, }); if (requestNumber.current !== currentRequest) return; if (outcome.status === 'rejected') { @@ -79,7 +85,7 @@ export function TorchInstallPreview({ tag, onBack, onInstall }: TorchInstallPrev } const selection = outcome.preview; if (!selection.previewId || selection.expiresInSeconds <= 0 || selection.tag !== tag - || selection.build !== build || selection.adapter !== 'none') { + || selection.build !== build || selection.adapter !== selectedAdapter) { setError('Torch selection could not be confirmed. Try again.'); setStarting(false); return; @@ -111,6 +117,19 @@ export function TorchInstallPreview({ tag, onBack, onInstall }: TorchInstallPrev This flow uses official binary wheels only. Device use, image generation, and socket startup need later runtime checks.

+ {runtimeOptions?.adapters.includes('flux2') && ( +
+ Image adapter + + +
+ )} {loadingOptions &&

} {starting &&

} {error &&

{error}

} From b1ff6b5401980628bab03c207879fd663167d825 Mon Sep 17 00:00:00 2001 From: MrScripty Date: Sat, 26 Sep 2026 20:05:34 -0700 Subject: [PATCH 2/7] Record isolated Torch 2.14 FLUX qualification A managed v2.14.0+cu132 FLUX install passed CUDA, adapter import, and owned profile startup checks in an isolated launcher root. Real model admission stopped at the existing 42 GiB available RAM gate, so no image or Tuldok claim is made. --- .../upstream-version-manager-progress.md | 22 +++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md index b4c9e58a..26deb2f9 100644 --- a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md +++ b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md @@ -78,6 +78,28 @@ the Debian package SHA-256 is `b010a8e86cddd5c07cd19e28da56b0c09f9e5f62613af5d6b82bd1192fe574b3`. Neither package was used for a v2.14+FLUX install or real Tuldok generation. +An isolated follow-up under +`launcher-data/cache/torch-qualification/v214-flux2-e2e-root/` used the real +`preview_torch_runtime` and `install_version` RPC flow for +`v2.14.0`/`cu132`/Python `auto`/`flux2`. The first attempt was cancelled before +completion to avoid a redundant large-wheel transfer from an empty isolated +cache; a private copy of the existing managed pip cache was used for the second +attempt. That attempt installed 67 resolved, hashed artifacts and completed +successfully. Its managed Python 3.14.7 probe passed exact Torch +`2.14.0+cu132` import, CPU and CUDA operations, and FLUX adapter imports +(Diffusers 0.37.0, Transformers 4.57.6). The adapter probe is correctly +`inconclusive` until a model loads. The isolated runtime was explicitly selected, +and its owned Pumas Torch profile passed the protocol-3 startup trial. + +The isolated `serve_model` call reached the existing 42 GiB FLUX RAM admission +gate and returned `insufficient_memory`; Pumas measured 39.16 GiB available RAM +and 23.17 GiB free GPU memory. No model was loaded or advertised by +`/v1/models`. The isolated profile and RPC server were stopped. The exact +resolution, lock, probe, and install log remain in that qualification root. +This confirms the managed v2.14+FLUX dependency and startup path but leaves +Pumas image generation and Tuldok display/save untested until the RAM gate can +pass. The main selected `v2.14.0` installation remains the earlier Core recipe. + ## 2026-09-26 — Bounded live transfer and packaged code check The latest successful `install-v2.14.0-1790470772611.log` used cached wheels, From 1a72ce811bca270238f1e362a2fc648ee4f9cfe5 Mon Sep 17 00:00:00 2001 From: MrScripty Date: Sat, 26 Sep 2026 22:10:11 -0700 Subject: [PATCH 3/7] Qualify Torch 2.14 FLUX Klein serving through Pumas and Tuldok --- docs/contracts/image-generation.md | 8 +- .../upstream-v214-cu132-flux2-klein-e2e.md | 37 ++ .../upstream-version-manager-progress.md | 32 +- .../pumas-rpc/src/handlers/serving_torch.rs | 236 ++++++++- scripts/acceptance/flux2_memory_probe.py | 227 ++++++++ .../acceptance/flux2_v214_rpc_acceptance.py | 487 ++++++++++++++++++ 6 files changed, 998 insertions(+), 29 deletions(-) create mode 100644 docs/plans/torch-diffusion-serving/reports/upstream-v214-cu132-flux2-klein-e2e.md create mode 100755 scripts/acceptance/flux2_memory_probe.py create mode 100644 scripts/acceptance/flux2_v214_rpc_acceptance.py diff --git a/docs/contracts/image-generation.md b/docs/contracts/image-generation.md index 4cdb2fd8..c3bf3d16 100644 --- a/docs/contracts/image-generation.md +++ b/docs/contracts/image-generation.md @@ -101,5 +101,9 @@ owns publication of the loaded model. The Klein adapter uses the local `flux-2-klein-9b-kv-fp8.safetensors` checkpoint, one library-managed `Qwen/Qwen3-8B` package and the standalone `split_files/vae/flux2-vae.safetensors` from `Comfy-Org/flux2-dev`. It requires -42 GiB available system RAM according to Pumas telemetry. Acquire these components -through Pumas; the adapter neither downloads them nor upgrades its dependencies. +38 GiB available system RAM according to Pumas telemetry when the selected +library artifact is FP8 and its Qwen3 config declares matching 128×128 block +FP8 weights. BF16 or unrecognized encoder artifacts require 42 GiB. These are +admission checks; generation still depends on available host and GPU memory. +Acquire these components through Pumas; the adapter neither downloads them nor +upgrades its dependencies. diff --git a/docs/plans/torch-diffusion-serving/reports/upstream-v214-cu132-flux2-klein-e2e.md b/docs/plans/torch-diffusion-serving/reports/upstream-v214-cu132-flux2-klein-e2e.md new file mode 100644 index 00000000..ac7841cb --- /dev/null +++ b/docs/plans/torch-diffusion-serving/reports/upstream-v214-cu132-flux2-klein-e2e.md @@ -0,0 +1,37 @@ +# Torch 2.14 CUDA FLUX.2 Klein and Tuldok acceptance + +Date: 2026-09-26 America/Vancouver. Pumas source: `work/torch-version-management`, after `b1ff6b54` and the FP8-specific admission repair in this change. This trial used the isolated launcher root `launcher-data/cache/torch-qualification/v214-flux2-e2e-root/`, not the main desktop launcher root. Its managed runtime is upstream Torch `2.14.0+cu132`, managed Python 3.14.7, adapter `flux2`, Diffusers 0.37.0 and Transformers 4.57.6. The selected image profile was `torch-image-acceptance`. + +## Why the checkpoint size was misleading + +The FLUX.2 Klein checkpoint is 9,818,935,984 bytes (9.14 GiB) on disk. The installed adapter expands its scaled FP8 transformer weights to BF16 and also loads the Qwen3-8B encoder and standalone VAE. Pumas selected the library-managed FP8 Qwen3 package, whose config declares `quant_method: fp8` and 128×128 blocks. The previous unconditional 42 GiB available-RAM admission rejected this host before model load at about 39 GiB available. The repaired admission permits 38 GiB only when both the selected library record identifies FP8 and the safely resolved Qwen3 config matches. BF16 and unknown variants retain 42 GiB. + +## Real bounded memory and image runs + +All runs used the installed runtime and real library assets, no mocked progress or image data. The native probe is `scripts/acceptance/flux2_memory_probe.py`; the managed acceptance runner is `scripts/acceptance/flux2_v214_rpc_acceptance.py`. They require a systemd cgroup memory limit and disabled swap. The managed runner starts a fresh Pumas RPC on port zero, verifies model readiness in `/v1/models`, calls `/v1/images/generations`, runs the actual Tuldok browser test when requested, and unloads/stops task-owned processes. + +| Run | Result | Cgroup peak | Limit pressure / OOM | +| --- | --- | ---: | --- | +| Native 512×512 | PNG, seed 42, four steps; SHA-256 `8351e2b416e5651d64039238a76426e4c777c054b92511648894f7799b0fcb60` | 27,054,182,400 B | 0 / 0 at 32 GiB cap | +| Native 1280×720 | PNG, seed 42; SHA-256 `dd8c7b98590851c4595f56ce432658e88beca8832d47897f1f03528601e65578` | 34,359,738,368 B | 74 / 0 at 32 GiB cap | +| Managed RPC/gateway 512×512 | Loaded and advertised image model; endpoint returned the same SHA-256 as native; unload and profile stop passed | 30,644,625,408 B | 0 / 0 at 32 GiB cap | +| Managed RPC/gateway plus Tuldok browser | 512×512 gateway request passed; Tuldok discovered, generated, displayed, saved and imported a real 1280×720 PNG | 36,507,394,048 B | 490 / 0 at 34 GiB cap | + +The 1280×720 runs touched their cgroup caps and incurred memory reclaim; their peaks are **capped observations**, not uncapped workload maxima. There were no cgroup OOM or OOM-kill events. During the full Tuldok trial, the active guard measured at least 22,444,679,168 bytes (20.9 GiB) of host `MemAvailable`; the RPC and Torch sidecar were verified in the same bounded scope. The 38 GiB Pumas check is an admission heuristic for this qualified FP8 encoder path, not a memory reservation or guarantee for arbitrary image dimensions. + +The real Tuldok browser requested 1280×720, seed 42, four steps and guidance 1. It reported 13.995 seconds for the browser job, displayed the decoded image at 1280×720, saved its PNG, and automatically added it to the collection. Saved PNG SHA-256: `ba32e3506ebee5a9e48c35c506777fa6b946b3ea9d7c45c86d1b6befd454750f`. Visual inspection shows the requested red teapot and yellow lemon on a blue table; the image also contains a second red vessel. Retained evidence is under `launcher-data/cache/torch-qualification/v214-flux2-e2e-root/tuldok-flux2-v214-real/{result.json,saved.png,display.png,browser.log}`. This evidence directory is local and ignored by Git. + +Five focused Rust admission tests, Rust formatting, a rebuilt debug RPC binary, syntax checks for both acceptance scripts, the native image runs, managed gateway runs, and real Tuldok browser acceptance passed. Independent read-only review found no remaining blocker in the FP8 classifier or test cleanup. All task-owned RPC, sidecar and browser processes stopped after the runs. + +## Desktop deployment state + +The main selected `v2.14.0` runtime remains the earlier Core (`adapter: none`) +installation; it cannot serve images until replaced through the managed +installer with a FLUX.2-enabled recipe. A new local Linux AppImage and deb were +built with the repaired release RPC. The AppImage SHA-256 is +`b05e198d0a29ec5049b7afaaa9678afd41bde861e7ce34d5f7e27fabdc60711c`; +the deb SHA-256 is +`af8d0e8ae81e087806218aa254a1a28a7be656d75a04c0211c22bd7b8d0f17c6`. +Both extracted packages matched their staged RPC/resources and passed RPC +`/health` smoke tests. The packaged desktop install/serve/Tuldok flow still +needs acceptance after the main runtime is replaced. Nothing was published. diff --git a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md index 26deb2f9..7bb8d6ae 100644 --- a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md +++ b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md @@ -1,11 +1,39 @@ # Upstream Torch version manager progress Date: 2026-09-26 (America/Vancouver). This is an implementation and evidence -inventory, not a completed Pumas desktop acceptance. The prior +inventory. Torch 2.14 FLUX.2 Klein now has isolated real RPC/gateway and Tuldok +browser acceptance; the main selected desktop runtime still has the Core recipe. +The prior [A1 packaged acceptance](a1-packaged-acceptance.md) applies only to its original `v2.9.1` / CPython 3.12 / CUDA 13.0 / Linux x86_64 combination. The Tuldok image evidence below uses a separate, exact Torch 2.10 tuple and does not qualify a -newly resolved upstream release. +newly resolved upstream release. The isolated Torch 2.14 result is recorded below. + +## 2026-09-26 — Real Torch 2.14 FLUX.2 Klein endpoint and Tuldok acceptance + +The [exact v2.14/cu132/FLUX.2 report](upstream-v214-cu132-flux2-klein-e2e.md) +records a real isolated managed runtime with the library's FLUX.2 Klein 9B KV +FP8 checkpoint, FP8 Qwen3-8B encoder, and standalone VAE. Although the +checkpoint is 9.14 GiB on disk, the adapter expands transformer weights to +BF16 and the first high-resolution image touched a 32 GiB test limit. A native +512×512 and 1280×720 PNG both generated. The previous 42 GiB admission was +repaired to permit 38 GiB only for a selected validated FP8 Qwen artifact with +matching in-package Qwen3 FP8 config; BF16 and unknown variants retain 42 GiB. + +The rebuilt source RPC loaded FLUX in a fresh isolated launcher root and +advertised it in `/v1/models`. Real `/v1/images/generations` requests returned +decoded PNGs. The actual Tuldok browser discovered the image model, generated +at 1280×720, displayed the image, saved it, and imported it into its collection. +The full run used a 34 GiB capped scope with no OOM events; 490 memory-limit +reclaim events show pressure at the cap. The host stayed at least 20.9 GiB +available during that run. The model, profile, RPC and browser were stopped. + +This qualifies the isolated v2.14+FLUX tuple and the Pumas endpoint/Tuldok path. +The main selected v2.14 installation is still Core (`adapter: none`) and cannot +serve images. It needs a managed FLUX recipe replacement for desktop use. New +local, unpublished AppImage/deb artifacts include this RPC change and passed +extracted resource and backend-health smokes; packaged desktop install/serve +and Tuldok acceptance remain open. No release was published. ## 2026-09-26 — Repair pip rate collection across versions and keep it visible diff --git a/rust/crates/pumas-rpc/src/handlers/serving_torch.rs b/rust/crates/pumas-rpc/src/handlers/serving_torch.rs index efdb7281..c366ff41 100644 --- a/rust/crates/pumas-rpc/src/handlers/serving_torch.rs +++ b/rust/crates/pumas-rpc/src/handlers/serving_torch.rs @@ -14,7 +14,7 @@ use pumas_library::models::{ UnserveModelRequest, UnserveModelResponse, }; use serde_json::Value; -use std::path::PathBuf; +use std::path::{Path, PathBuf}; const NUNCHAKU_REPOSITORIES: &[&str] = &[ "nunchaku-ai/nunchaku-z-image-turbo", @@ -24,6 +24,12 @@ const Z_IMAGE_COMPONENTS_REPO: &str = "Tongyi-MAI/Z-Image-Turbo"; const FLUX_REPOSITORY: &str = "black-forest-labs/FLUX.2-klein-9b-kv-fp8"; const FLUX_CHECKPOINT: &str = "flux-2-klein-9b-kv-fp8.safetensors"; const NUNCHAKU_CHECKPOINT: &str = "svdq-fp4_r128-z-image-turbo.safetensors"; +const GIB: u64 = 1024 * 1024 * 1024; + +struct ResolvedComponents { + path: String, + selected_fp8: bool, +} pub(super) async fn serve_torch_model( state: &AppState, @@ -66,7 +72,7 @@ pub(super) async fn serve_torch_model( ) .await?; let vae = if is_flux { - Some(resolve_components(state, "Comfy-Org/flux2-dev").await?) + Some(resolve_components(state, "Comfy-Org/flux2-dev").await?.path) } else { None }; @@ -108,11 +114,23 @@ pub(super) async fn serve_torch_model( }; let resources = state.api.get_system_resources().await?; if is_flux { - let ram = &resources.resources.ram; - let available = ram.total as f64 * (1.0 - f64::from(ram.usage) / 100.0); - if !resources.success || available < (42_u64 * 1024 * 1024 * 1024) as f64 { - return non_critical_failure_response(state, fail(ModelServeErrorCode::InsufficientMemory, - "Klein's scaled FP8 to BF16 CPU-offload policy requires 42 GiB available system RAM in Pumas telemetry")).await; + let fp8_encoder = is_validated_fp8_qwen3_encoder(&components).await; + if !flux_ram_sufficient( + resources.success, + resources.resources.ram.total, + resources.resources.ram.usage, + fp8_encoder, + ) { + let message = if fp8_encoder { + "Klein's FP8 CPU-offload policy requires 38 GiB available system RAM in Pumas telemetry" + } else { + "Klein's CPU-offload policy requires 42 GiB available system RAM in Pumas telemetry" + }; + return non_critical_failure_response( + state, + fail(ModelServeErrorCode::InsufficientMemory, message), + ) + .await; } } let gpu = resources.resources.gpu; @@ -262,7 +280,7 @@ pub(super) async fn serve_torch_model( "nunchaku-z-image-turbo" }, &ImageModelComponents { - pipeline_path: &components, + pipeline_path: &components.path, vae_path: vae.as_deref(), }, ) @@ -365,8 +383,61 @@ pub(super) async fn serve_torch_model( })?) } +fn flux_ram_sufficient(telemetry_success: bool, total: u64, usage: f32, fp8_encoder: bool) -> bool { + if !telemetry_success || total == 0 || !usage.is_finite() || !(0.0..=100.0).contains(&usage) { + return false; + } + let required_gib = if fp8_encoder { 38 } else { 42 }; + let available = total as f64 * (1.0 - f64::from(usage) / 100.0); + available >= (required_gib * GIB) as f64 +} + +fn is_fp8_qwen3_config(config: &Value) -> bool { + config.get("model_type").and_then(Value::as_str) == Some("qwen3") + && config.get("hidden_size").and_then(Value::as_u64) == Some(4096) + && config + .pointer("/quantization_config/quant_method") + .and_then(Value::as_str) + == Some("fp8") + && config + .pointer("/quantization_config/weight_block_size") + .and_then(Value::as_array) + .is_some_and(|block| { + block.len() == 2 && block.iter().all(|size| size.as_u64() == Some(128)) + }) +} + +async fn is_validated_fp8_qwen3_encoder(components: &ResolvedComponents) -> bool { + if !components.selected_fp8 { + return false; + } + let package = Path::new(&components.path); + // The component resolver already validated and canonicalized this package. + // Canonicalize the config as well so a symlink cannot escape the package. + let Ok(config_path) = tokio::fs::canonicalize(package.join("config.json")).await else { + return false; + }; + if !config_path.starts_with(package) { + return false; + } + let Ok(contents) = tokio::fs::read(config_path).await else { + return false; + }; + serde_json::from_slice::(&contents).is_ok_and(|config| is_fp8_qwen3_config(&config)) +} + +fn selected_artifact_is_fp8(metadata: &Value) -> bool { + metadata + .get("selected_artifact_quant") + .and_then(Value::as_str) + == Some("FP8") +} + /// Resolve a single validated package through the existing library authority. -async fn resolve_components(state: &AppState, repository: &str) -> pumas_library::Result { +async fn resolve_components( + state: &AppState, + repository: &str, +) -> pumas_library::Result { let mut candidates: Vec<_> = state .api .model_library() @@ -378,21 +449,11 @@ async fn resolve_components(state: &AppState, repository: &str) -> pumas_library // Retain the BF16 source while preferring the library's converted encoder. // Ambiguous variants still require an explicit library correction below. if repository == "Qwen/Qwen3-8B" - && candidates.iter().any(|model| { - model - .metadata - .get("selected_artifact_quant") - .and_then(Value::as_str) - == Some("FP8") - }) + && candidates + .iter() + .any(|model| selected_artifact_is_fp8(&model.metadata)) { - candidates.retain(|model| { - model - .metadata - .get("selected_artifact_quant") - .and_then(Value::as_str) - == Some("FP8") - }); + candidates.retain(|model| selected_artifact_is_fp8(&model.metadata)); } let [model] = candidates.as_slice() else { return Err(pumas_library::PumasError::InvalidParams { @@ -433,9 +494,15 @@ async fn resolve_components(state: &AppState, repository: &str) -> pumas_library message: "FLUX.2 VAE is outside its validated component package".into(), }); } - return Ok(vae.to_string_lossy().into_owned()); + return Ok(ResolvedComponents { + path: vae.to_string_lossy().into_owned(), + selected_fp8: false, + }); } - Ok(package.to_string_lossy().into_owned()) + Ok(ResolvedComponents { + path: package.to_string_lossy().into_owned(), + selected_fp8: repository == "Qwen/Qwen3-8B" && selected_artifact_is_fp8(&model.metadata), + }) } pub(super) async fn unserve_torch_model( @@ -485,3 +552,122 @@ pub(super) async fn unserve_torch_model( snapshot: Some(snapshot), })?) } + +#[cfg(test)] +mod tests { + use super::*; + use serde_json::json; + + #[tokio::test] + async fn fp8_requires_matching_config_in_resolved_package() { + let package = tempfile::tempdir().unwrap(); + let config_path = package.path().join("config.json"); + let components = ResolvedComponents { + path: package.path().to_string_lossy().into_owned(), + selected_fp8: true, + }; + let fp8 = json!({ + "model_type": "qwen3", + "hidden_size": 4096, + "quantization_config": { + "quant_method": "fp8", + "weight_block_size": [128, 128] + } + }); + tokio::fs::write(&config_path, fp8.to_string()) + .await + .unwrap(); + assert!(is_validated_fp8_qwen3_encoder(&components).await); + + for changed in [ + json!({"model_type": "qwen2"}), + json!({"hidden_size": 8192}), + json!({"quantization_config": null}), + json!({"quantization_config": {"quant_method": "bitsandbytes", "weight_block_size": [128, 128]}}), + json!({"quantization_config": {"quant_method": "fp8", "weight_block_size": [64, 128]}}), + json!({"quantization_config": {"quant_method": "fp8"}}), + ] { + let mut config = fp8.clone(); + for (key, value) in changed.as_object().unwrap() { + config[key] = value.clone(); + } + tokio::fs::write(&config_path, config.to_string()) + .await + .unwrap(); + assert!(!is_validated_fp8_qwen3_encoder(&components).await); + } + tokio::fs::write(&config_path, "invalid json") + .await + .unwrap(); + assert!(!is_validated_fp8_qwen3_encoder(&components).await); + tokio::fs::remove_file(&config_path).await.unwrap(); + assert!(!is_validated_fp8_qwen3_encoder(&components).await); + } + + #[tokio::test] + async fn fp8_looking_config_with_bf16_or_unknown_catalog_selection_keeps_42_gib() { + let package = tempfile::tempdir().unwrap(); + let config = json!({ + "model_type": "qwen3", "hidden_size": 4096, + "quantization_config": {"quant_method": "fp8", "weight_block_size": [128, 128]} + }); + tokio::fs::write(package.path().join("config.json"), config.to_string()) + .await + .unwrap(); + for metadata in [ + json!({"selected_artifact_quant": "BF16"}), + json!({"selected_artifact_quant": "UNKNOWN"}), + json!({}), + ] { + let components = ResolvedComponents { + path: package.path().to_string_lossy().into_owned(), + selected_fp8: selected_artifact_is_fp8(&metadata), + }; + let fp8_encoder = is_validated_fp8_qwen3_encoder(&components).await; + assert!(!fp8_encoder); + assert!(!flux_ram_sufficient(true, 100 * GIB, 62.0, fp8_encoder)); + } + } + + #[cfg(unix)] + #[tokio::test] + async fn fp8_config_symlink_cannot_escape_package() { + let package = tempfile::tempdir().unwrap(); + let outside = tempfile::tempdir().unwrap(); + let components = ResolvedComponents { + path: package.path().to_string_lossy().into_owned(), + selected_fp8: true, + }; + let config = json!({ + "model_type": "qwen3", "hidden_size": 4096, + "quantization_config": {"quant_method": "fp8", "weight_block_size": [128, 128]} + }); + std::fs::write(outside.path().join("config.json"), config.to_string()).unwrap(); + std::os::unix::fs::symlink( + outside.path().join("config.json"), + package.path().join("config.json"), + ) + .unwrap(); + assert!(!is_validated_fp8_qwen3_encoder(&components).await); + } + + #[test] + fn flux_ram_threshold_depends_on_verified_fp8_encoder() { + let total = 100 * GIB; + assert!(flux_ram_sufficient(true, total, 62.0, true)); + assert!(!flux_ram_sufficient(true, total, 62.01, true)); + assert!(!flux_ram_sufficient(true, total, 62.0, false)); + assert!(flux_ram_sufficient(true, total, 58.0, false)); + assert!(!flux_ram_sufficient(true, total, 58.01, false)); + } + + #[test] + fn flux_ram_rejects_failed_or_invalid_telemetry() { + let total = 100 * GIB; + assert!(!flux_ram_sufficient(false, total, 0.0, true)); + assert!(!flux_ram_sufficient(true, 0, 0.0, true)); + assert!(!flux_ram_sufficient(true, total, f32::NAN, true)); + assert!(!flux_ram_sufficient(true, total, -1.0, true)); + assert!(!flux_ram_sufficient(true, total, 101.0, true)); + } +} diff --git a/scripts/acceptance/flux2_memory_probe.py b/scripts/acceptance/flux2_memory_probe.py new file mode 100755 index 00000000..bf40cdd8 --- /dev/null +++ b/scripts/acceptance/flux2_memory_probe.py @@ -0,0 +1,227 @@ +#!/usr/bin/env python3 +"""Measure a real FLUX.2 Klein load and image generation in the isolated Torch v2.14 runtime. + +Run from the repository root, for example: + + systemd-run --user --scope -p MemoryMax=32G -p MemorySwapMax=0 \ + -- scripts/acceptance/flux2_memory_probe.py --output /tmp/flux2-memory.png + +The script re-execs itself with the qualified v2.14 virtualenv's Python. It +prints one JSON record per phase and a final PNG SHA-256. The systemd scope is +the hard memory bound; this process also has a 15-minute wall-clock timeout. +""" + +import argparse +import hashlib +import json +import os +from pathlib import Path +import signal +import sys +import threading +import time +import traceback + + +ROOT = Path(__file__).resolve().parents[2] +RUNTIME = ROOT / "launcher-data/cache/torch-qualification/v214-flux2-e2e-root/torch-versions/v2.14.0" +PYTHON = RUNTIME / "venv/bin/python" +DEFAULT_CHECKPOINT = ROOT / "shared-resources/models/diffusion/black-forest-labs/flux_2-klein-9b-kv-fp8/flux-2-klein-9b-kv-fp8.safetensors" +DEFAULT_ENCODER = ROOT / "shared-resources/models/llm/qwen3/qwen--qwen3-8b__files_97c4978527af-fp8" +DEFAULT_VAE = ROOT / "shared-resources/models/diffusion/flux2/comfy-org--flux2-dev__files_7432f8b613b4/split_files/vae/flux2-vae.safetensors" + + +def emit(phase, **fields): + print(json.dumps({"time": time.time(), "phase": phase, **fields}, sort_keys=True), flush=True) + + +def number(path): + try: + value = Path(path).read_text().strip() + return None if value == "max" else int(value) + except (OSError, ValueError): + return None + + +def memory_events(path): + try: + return {key: int(value) for key, value in (line.split() for line in Path(path).read_text().splitlines())} + except (OSError, ValueError): + return None + + +def cgroup_memory(): + try: + line = next(line for line in Path("/proc/self/cgroup").read_text().splitlines() if line.startswith("0::")) + group = line.split("::", 1)[1].lstrip("/") + base = Path("/sys/fs/cgroup") / group + return base + except (OSError, StopIteration): + return None + + +def rss_bytes(): + try: + for line in Path("/proc/self/status").read_text().splitlines(): + if line.startswith("VmRSS:"): + return int(line.split()[1]) * 1024 + except (OSError, ValueError, IndexError): + pass + return None + + +def mem_available_bytes(): + try: + for line in Path("/proc/meminfo").read_text().splitlines(): + if line.startswith("MemAvailable:"): + return int(line.split()[1]) * 1024 + except (OSError, ValueError, IndexError): + pass + return None + + +class MemorySampler: + def __init__(self): + self.group = cgroup_memory() + self.stop = threading.Event() + self.peak_rss = 0 + self.peak_cgroup = 0 + self.thread = threading.Thread(target=self._sample, daemon=True) + + def _sample(self): + while not self.stop.is_set(): + self.peak_rss = max(self.peak_rss, rss_bytes() or 0) + if self.group: + self.peak_cgroup = max(self.peak_cgroup, number(self.group / "memory.current") or 0) + self.stop.wait(0.2) + + def start(self): + self.thread.start() + + def finish(self): + self.stop.set() + self.thread.join(timeout=1) + + def snapshot(self, torch=None): + data = { + "rss_bytes": rss_bytes(), + "sampled_peak_rss_bytes": self.peak_rss, + "host_mem_available_bytes": mem_available_bytes(), + } + if self.group: + data.update( + cgroup_current_bytes=number(self.group / "memory.current"), + cgroup_peak_bytes=number(self.group / "memory.peak"), + sampled_peak_cgroup_bytes=self.peak_cgroup, + cgroup_limit_bytes=number(self.group / "memory.max"), + cgroup_swap_limit_bytes=number(self.group / "memory.swap.max"), + cgroup_swap_current_bytes=number(self.group / "memory.swap.current"), + cgroup_events=memory_events(self.group / "memory.events"), + ) + if torch is not None and torch.cuda.is_available(): + free, total = torch.cuda.mem_get_info() + data.update( + cuda_free_bytes=free, + cuda_total_bytes=total, + cuda_allocated_bytes=torch.cuda.memory_allocated(), + cuda_reserved_bytes=torch.cuda.memory_reserved(), + cuda_peak_allocated_bytes=torch.cuda.max_memory_allocated(), + cuda_peak_reserved_bytes=torch.cuda.max_memory_reserved(), + ) + return data + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--checkpoint", type=Path, default=DEFAULT_CHECKPOINT) + parser.add_argument("--encoder", type=Path, default=DEFAULT_ENCODER) + parser.add_argument("--vae", type=Path, default=DEFAULT_VAE) + parser.add_argument("--output", type=Path, default=Path("/tmp/flux2-memory.png")) + parser.add_argument("--prompt", default="A red ceramic teapot beside a yellow lemon on a blue table") + parser.add_argument("--width", type=int, default=512) + parser.add_argument("--height", type=int, default=512) + parser.add_argument("--seed", type=int, default=42) + parser.add_argument("--timeout-seconds", type=int, default=900) + parser.add_argument("--max-memory-gib", type=int, default=32, help="largest permitted cgroup memory.max in GiB") + args = parser.parse_args() + + if Path(sys.prefix).resolve() != (RUNTIME / "venv").resolve(): + if not PYTHON.is_file(): + parser.error(f"isolated Torch interpreter missing: {PYTHON}") + os.execv(str(PYTHON), [str(PYTHON), str(Path(__file__).resolve()), *sys.argv[1:]]) + + if args.width < 64 or args.height < 64 or args.width > 2048 or args.height > 2048 or args.width % 16 or args.height % 16: + parser.error("width and height must be multiples of 16 from 64 through 2048") + if not 1 <= args.timeout_seconds <= 3600: + parser.error("timeout must be from 1 to 3600 seconds") + if not 1 <= args.max_memory_gib <= 64: + parser.error("max-memory-gib must be from 1 to 64") + for label, path in (("checkpoint", args.checkpoint), ("vae", args.vae)): + if not path.is_file(): + parser.error(f"{label} missing: {path}") + for name in ("config.json", "model.safetensors.index.json", "tokenizer.json"): + if not (args.encoder / name).is_file(): + parser.error(f"encoder component missing: {args.encoder / name}") + config = json.loads((args.encoder / "config.json").read_text()) + if config.get("model_type") != "qwen3" or config.get("hidden_size") != 4096: + parser.error("encoder must be the validated Qwen3-8B component") + if config.get("quantization_config", {}).get("quant_method") != "fp8": + parser.error("encoder must be the Pumas FP8 Qwen3-8B component") + + group = cgroup_memory() + limit = number(group / "memory.max") if group else None + swap_limit = number(group / "memory.swap.max") if group else None + if limit is None or limit > args.max_memory_gib * 1024**3 or swap_limit != 0: + parser.error( + f"run in a bounded cgroup with memory.max <= {args.max_memory_gib} GiB " + f"and memory.swap.max = 0 (observed memory.max={limit}, swap.max={swap_limit})" + ) + for name in ("flux2.py", "diffusion.py"): + if not (RUNTIME / name).is_file(): + parser.error(f"isolated Torch adapter missing: {RUNTIME / name}") + + os.environ.update(HF_HUB_OFFLINE="1", TRANSFORMERS_OFFLINE="1", DIFFUSERS_OFFLINE="1") + sys.path.insert(0, str(RUNTIME)) + sampler = MemorySampler() + sampler.start() + started = time.monotonic() + torch = None + + def timeout(_signum, _frame): + raise TimeoutError(f"probe exceeded {args.timeout_seconds} seconds") + + signal.signal(signal.SIGALRM, timeout) + signal.alarm(args.timeout_seconds) + try: + emit("start", runtime=str(RUNTIME), checkpoint=str(args.checkpoint), encoder=str(args.encoder), encoder_quantization_config=config["quantization_config"], vae=str(args.vae), output=str(args.output), memory=sampler.snapshot()) + import torch as torch_module + torch = torch_module + from flux2 import Flux2Klein + + if not torch.cuda.is_available() or torch.cuda.get_device_capability(0) != (12, 0): + raise RuntimeError("qualified sm_120 CUDA device 0 is unavailable") + emit("imports_complete", torch_version=torch.__version__, memory=sampler.snapshot(torch)) + load_start = time.monotonic() + emit("load_start", memory=sampler.snapshot(torch)) + adapter = Flux2Klein(str(args.checkpoint), str(args.encoder), str(args.vae), torch.device("cuda:0")) + emit("load_complete", elapsed_seconds=round(time.monotonic() - load_start, 3), memory=sampler.snapshot(torch)) + + generation_start = time.monotonic() + emit("generate_start", width=args.width, height=args.height, steps=adapter.steps, seed=args.seed, memory=sampler.snapshot(torch)) + image = adapter.generate(args.prompt, args.width, args.height, args.seed, threading.Event()) + emit("generate_complete", elapsed_seconds=round(time.monotonic() - generation_start, 3), memory=sampler.snapshot(torch)) + args.output.parent.mkdir(parents=True, exist_ok=True) + image.save(args.output, format="PNG") + digest = hashlib.sha256(args.output.read_bytes()).hexdigest() + emit("complete", elapsed_seconds=round(time.monotonic() - started, 3), png=str(args.output), png_sha256=digest, png_bytes=args.output.stat().st_size, memory=sampler.snapshot(torch)) + except BaseException as error: + emit("failed", error_type=type(error).__name__, error=str(error), elapsed_seconds=round(time.monotonic() - started, 3), memory=sampler.snapshot(torch)) + traceback.print_exc() + raise SystemExit(1) from error + finally: + signal.alarm(0) + sampler.finish() + + +if __name__ == "__main__": + main() diff --git a/scripts/acceptance/flux2_v214_rpc_acceptance.py b/scripts/acceptance/flux2_v214_rpc_acceptance.py new file mode 100644 index 00000000..7beb43c9 --- /dev/null +++ b/scripts/acceptance/flux2_v214_rpc_acceptance.py @@ -0,0 +1,487 @@ +#!/usr/bin/env python3 +"""Exercise FLUX.2 through the real managed Pumas RPC and image gateway. + +Run from any directory inside a bounded scope, for example: + + systemd-run --user --scope -p MemoryMax=32G -p MemorySwapMax=0 \ + -- python3 scripts/acceptance/flux2_v214_rpc_acceptance.py \ + --output /tmp/flux2-v214-rpc.png + +The default image is 512x512. Pass --width 1280 --height 720 for the full-size +trial when the memory result supports it. + +The existing isolated launcher root must already contain selected Torch v2.14.0, +the managed torch-image-acceptance profile, and linked library model assets. +Only that root is used for Pumas state. The RPC log is saved beside the PNG. +""" + +import argparse +import base64 +import hashlib +import json +import os +from pathlib import Path +import re +import signal +import struct +import subprocess +import sys +import threading +import time +import urllib.error +import urllib.request + + +REPO = Path(__file__).resolve().parents[2] +ROOT = REPO / "launcher-data/cache/torch-qualification/v214-flux2-e2e-root" +BINARY = REPO / "rust/target/debug/pumas-rpc" +MODEL_ID = "diffusion/black-forest-labs/flux_2-klein-9b-kv-fp8" +PROFILE_ID = "torch-image-acceptance" +ALIAS = MODEL_ID +TULDOK_TEST = REPO.parent.parent / "creative-media/Tuldok/tests/browser_images_real.cjs" +PROMPT = "A red ceramic teapot beside a yellow lemon on a blue table" +SEED = 42 +HTTP = urllib.request.build_opener(urllib.request.ProxyHandler({})) + + +def emit(phase, **fields): + print(json.dumps({"phase": phase, **fields}, sort_keys=True), flush=True) + + +def require(condition, label, value=None): + if not condition: + raise RuntimeError(f"{label}: {value!r}") + + +def cgroup_file(name): + try: + group = next(line.split("::", 1)[1].lstrip("/") for line in + Path("/proc/self/cgroup").read_text().splitlines() + if line.startswith("0::")) + return Path("/sys/fs/cgroup") / group / name + except (OSError, StopIteration): + return None + + +def cgroup_value(name): + path = cgroup_file(name) + try: + return path.read_text().strip() if path else None + except OSError: + return None + + +def memory_events(): + value = cgroup_value("memory.events") + return dict((key, int(count)) for key, count in + (line.split() for line in value.splitlines())) if value else {} + + +def process_cgroup(pid): + try: + return next(line.split("::", 1)[1] for line in + Path(f"/proc/{pid}/cgroup").read_text().splitlines() + if line.startswith("0::")) + except (OSError, StopIteration): + return None + + +def child_processes(pid): + found = set() + pending = [pid] + while pending: + parent = pending.pop() + try: + children = [int(value) for value in + Path(f"/proc/{parent}/task/{parent}/children").read_text().split()] + except (OSError, ValueError): + continue + for child in children: + if child not in found: + found.add(child) + pending.append(child) + return found + + +def owned_group_members(group_id, group_cgroup): + """Return live members of a spawned process group still in this scope.""" + members = [] + for path in Path("/proc").iterdir(): + if not path.name.isdigit(): + continue + pid = int(path.name) + try: + stat = (path / "stat").read_text() + fields = stat.rsplit(") ", 1)[1].split() + state, process_group = fields[0], int(fields[2]) + except (OSError, IndexError, ValueError): + continue + if state != "Z" and process_group == group_id and process_cgroup(pid) == group_cgroup: + members.append(pid) + return members + + +def stop_owned_group(group_id, group_cgroup): + for sig in (signal.SIGTERM, signal.SIGKILL): + if not owned_group_members(group_id, group_cgroup): + return + try: + os.killpg(group_id, sig) + except ProcessLookupError: + return + deadline = time.monotonic() + (5 if sig == signal.SIGTERM else 2) + while time.monotonic() < deadline: + if not owned_group_members(group_id, group_cgroup): + return + time.sleep(0.1) + require(not owned_group_members(group_id, group_cgroup), + "owned browser process group survived cleanup", group_id) + + +def owned_process_cgroups(rpc_pid): + processes = {rpc_pid, *child_processes(rpc_pid)} + records = [] + for pid in sorted(processes): + try: + command = Path(f"/proc/{pid}/cmdline").read_bytes().replace(b"\0", b" ").decode(errors="replace") + except OSError: + command = "" + records.append({"pid": pid, "cgroup": process_cgroup(pid), "command": command[:300]}) + return records + + +def owned_sidecars(): + """Find only this scope's sidecar running from the isolated runtime.""" + runtime = (ROOT / "torch-versions/v2.14.0").resolve() + group = process_cgroup(os.getpid()) + found = [] + for path in Path("/proc").iterdir(): + if not path.name.isdigit(): + continue + pid = int(path.name) + try: + command = (path / "cmdline").read_bytes() + cwd = (path / "cwd").resolve(strict=True) + except OSError: + continue + if b"serve.py" in command.split(b"\0") and cwd == runtime and process_cgroup(pid) == group: + found.append({"pid": pid, "cgroup": group, "runtime_script": str(runtime / "serve.py")}) + return found + + +def stop_orphaned_sidecars(): + found = owned_sidecars() + for sidecar in found: + try: + os.kill(sidecar["pid"], signal.SIGTERM) + except ProcessLookupError: + pass + deadline = time.monotonic() + 5 + while found and time.monotonic() < deadline: + found = owned_sidecars() + if found: + time.sleep(0.1) + for sidecar in found: + try: + os.kill(sidecar["pid"], signal.SIGKILL) + except ProcessLookupError: + pass + require(not owned_sidecars(), "owned Torch sidecar remained after cleanup", found) + + +def host_available_bytes(): + try: + line = next(line for line in Path("/proc/meminfo").read_text().splitlines() + if line.startswith("MemAvailable:")) + return int(line.split()[1]) * 1024 + except (OSError, StopIteration, ValueError, IndexError): + return None + + +class HostMemoryGuard: + """Stop this runner's RPC tree and browser group if host headroom falls.""" + + def __init__(self, rpc_process): + self.rpc_process = rpc_process + self.browser_process = None + self.stop = threading.Event() + self.tripped = False + self.low_available = None + self.minimum_available = None + self.thread = threading.Thread(target=self.sample, daemon=True) + + def start(self): + self.thread.start() + + def finish(self): + self.stop.set() + self.thread.join(timeout=2) + + def sample(self): + while not self.stop.is_set(): + available = host_available_bytes() + if available is None or available < 6 * 1024**3: + self.tripped = True + self.low_available = available + browser = self.browser_process + if browser is not None: + try: + stop_owned_group(browser.pid, process_cgroup(os.getpid())) + except RuntimeError: + pass # Continue stopping the RPC and sidecar as well. + for pid in child_processes(self.rpc_process.pid) | { + sidecar["pid"] for sidecar in owned_sidecars() + }: + try: + os.kill(pid, signal.SIGKILL) + except ProcessLookupError: + pass + if self.rpc_process.poll() is None: + try: + os.killpg(self.rpc_process.pid, signal.SIGKILL) + except ProcessLookupError: + pass + return + self.minimum_available = min(self.minimum_available or available, available) + self.stop.wait(0.25) + + +def request_json(base, path, body=None, timeout=30): + data = None if body is None else json.dumps(body).encode("utf-8") + request = urllib.request.Request( + base + path, data=data, + headers={"Content-Type": "application/json"} if data is not None else {}, + ) + try: + with HTTP.open(request, timeout=timeout) as response: + return json.load(response) + except urllib.error.HTTPError as error: + detail = error.read(4096).decode("utf-8", errors="replace") + raise RuntimeError(f"{path}: HTTP {error.code}: {detail}") from error + + +def rpc(base, method, params=None, timeout=30): + reply = request_json(base, "/rpc", { + "jsonrpc": "2.0", "id": 1, "method": method, "params": params or {}, + }, timeout=timeout) + require("error" not in reply, f"RPC {method}", reply) + return reply["result"] + + +def await_rpc_port(process, log_path): + deadline = time.monotonic() + 60 + while time.monotonic() < deadline: + contents = log_path.read_text(errors="replace") + match = re.search(r"RPC_PORT=(\d+)", contents) + if match: + base = f"http://127.0.0.1:{match[1]}" + request_json(base, "/health", timeout=10) + return base + require(process.poll() is None, "RPC exited before port announcement", contents[-4000:]) + time.sleep(0.1) + raise RuntimeError(f"RPC did not announce a port: {log_path}") + + +def validate_png(data, expected_width, expected_height): + require(data.startswith(b"\x89PNG\r\n\x1a\n"), "image is not PNG") + require(len(data) >= 33 and data[12:16] == b"IHDR", "PNG lacks IHDR") + width, height = struct.unpack(">II", data[16:24]) + require((width, height) == (expected_width, expected_height), "PNG dimensions", (width, height)) + require(data[-12:-8] == b"\x00\x00\x00\x00" and data[-8:-4] == b"IEND", "PNG is incomplete") + return width, height + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--output", type=Path, default=Path("/tmp/flux2-v214-rpc.png")) + parser.add_argument("--width", type=int, default=512) + parser.add_argument("--height", type=int, default=512) + parser.add_argument("--tuldok", action="store_true", + help="after the gateway trial, run Tuldok's real 1280x720 browser acceptance (34 GiB scope)") + args = parser.parse_args() + require(64 <= args.width <= 2048 and args.width % 16 == 0 and + 64 <= args.height <= 2048 and args.height % 16 == 0, + "dimensions must be multiples of 16 from 64 through 2048", + (args.width, args.height)) + limit, swap = cgroup_value("memory.max"), cgroup_value("memory.swap.max") + max_gib = 34 if args.tuldok else 32 + try: + bounded = limit is not None and int(limit) <= max_gib * 1024**3 and swap == "0" + except ValueError: + bounded = False + require(bounded, f"run with memory.max <= {max_gib} GiB and memory.swap.max = 0", + {"limit": limit, "swap": swap}) + if args.tuldok: + available = host_available_bytes() + require(available is not None and available >= 6 * 1024**3, + "host MemAvailable must be at least 6 GiB", available) + require(BINARY.is_file(), "RPC binary missing", str(BINARY)) + require((ROOT / ".active-version-torch").read_text().strip() == "v2.14.0", "selected Torch version") + require((ROOT / "shared-resources/models").is_dir(), "linked model library missing") + profile_data = json.loads((ROOT / "launcher-data/metadata/runtime-profiles.json").read_text()) + profile = next((p for p in profile_data["profiles"] if p["profile_id"] == PROFILE_ID), None) + require(profile and profile.get("provider") == "torch" and + profile.get("provider_mode") == "torch_serve" and + profile.get("management_mode") == "managed" and + profile.get("enabled"), "selected managed Torch profile", profile) + + output = args.output.resolve() + output.parent.mkdir(parents=True, exist_ok=True) + log_path = output.with_suffix(".rpc.log") + env = {**os.environ, "XDG_CONFIG_HOME": str(ROOT / "config"), + "PUMAS_REGISTRY_DB_PATH": str(ROOT / "registry.db"), + "APPDATA": str(ROOT / "config"), + "HF_HUB_OFFLINE": "1", "TRANSFORMERS_OFFLINE": "1", "DIFFUSERS_OFFLINE": "1"} + loaded = False + base = None + process = None + guard = None + failure = None + events_before = memory_events() + require(bool(events_before) and cgroup_value("memory.peak") is not None, + "cgroup memory event and peak counters unavailable") + emit("start", cgroup_limit_bytes=int(limit), cgroup_swap_limit_bytes=int(swap), + cgroup_events_before=events_before, cgroup_path=process_cgroup(os.getpid()), + host_mem_available_bytes=host_available_bytes(), width=args.width, + height=args.height, seed=SEED) + with log_path.open("wb") as log: + process = subprocess.Popen( + [str(BINARY), "--launcher-root", str(ROOT), "--port", "0"], + cwd=ROOT, env=env, stdout=log, stderr=subprocess.STDOUT, + start_new_session=True, + ) + if args.tuldok: + guard = HostMemoryGuard(process) + guard.start() + try: + base = await_rpc_port(process, log_path) + emit("rpc_ready", endpoint=base, pid=process.pid, log=str(log_path)) + before = rpc(base, "get_serving_status") + require(not before.get("snapshot", {}).get("served_models"), "isolated root already serves models", before) + served = rpc(base, "serve_model", {"request": { + "model_id": MODEL_ID, + "config": {"provider": "torch", "profile_id": PROFILE_ID, + "device_mode": "gpu", "keep_loaded": True, "model_alias": ALIAS}, + }}, timeout=600) + require(served.get("success") and served.get("loaded"), "FLUX.2 load", served) + loaded = True + emit("model_loaded", model_id=MODEL_ID, alias=ALIAS, + owned_process_cgroups=owned_process_cgroups(process.pid), + sidecar_cgroups=owned_sidecars()) + models = request_json(base, "/v1/models", timeout=30) + require(any(item.get("id") == ALIAS and "image_generation" in item.get("capabilities", []) + for item in models.get("data", [])), "gateway model readiness", models) + emit("gateway_ready", endpoint=base + "/v1/images/generations") + result = request_json(base, "/v1/images/generations", { + "model": ALIAS, "prompt": PROMPT, "width": args.width, "height": args.height, + "seed": SEED, "n": 1, "response_format": "b64_json", + }, timeout=900) + images = result.get("data", []) + require(len(images) == 1 and isinstance(images[0].get("b64_json"), str), + "image gateway response", result) + png = base64.b64decode(images[0]["b64_json"], validate=True) + width, height = validate_png(png, args.width, args.height) + output.write_bytes(png) + emit("image_saved", path=str(output), width=width, height=height, + bytes=len(png), sha256=hashlib.sha256(png).hexdigest(), + metadata=result.get("metadata")) + if args.tuldok: + require(TULDOK_TEST.is_file(), "Tuldok browser acceptance missing", str(TULDOK_TEST)) + evidence = ROOT / "tuldok-flux2-v214-real" + evidence.mkdir(parents=True, exist_ok=True) + browser_log = evidence / "browser.log" + browser_env = {**env, "PUMAS_GATEWAY": base + "/v1", + "PUMAS_MODEL": MODEL_ID, + "TULDOK_EVIDENCE_DIR": str(evidence)} + emit("tuldok_start", evidence=str(evidence)) + with browser_log.open("wb") as log: + browser = subprocess.Popen( + ["node", str(TULDOK_TEST)], cwd=TULDOK_TEST.parent.parent, + env=browser_env, stdout=log, stderr=subprocess.STDOUT, + start_new_session=True, + ) + browser_group = browser.pid # start_new_session makes this the group ID. + browser_cgroup = process_cgroup(os.getpid()) + guard.browser_process = browser + try: + if browser.poll() is None: + require(os.getpgid(browser.pid) == browser_group and + process_cgroup(browser.pid) == browser_cgroup, + "browser process group ownership", browser_group) + browser.wait(timeout=900) + except subprocess.TimeoutExpired: + raise RuntimeError(f"Tuldok timed out; see {browser_log}") + finally: + # The browser test may be interrupted before its own finally runs. + stop_owned_group(browser_group, browser_cgroup) + browser.wait(timeout=10) + guard.browser_process = None + require(browser.returncode == 0, "Tuldok browser acceptance", str(browser_log)) + browser_result = json.loads((evidence / "result.json").read_text()) + require(browser_result.get("fixture") is False and + browser_result.get("model") == MODEL_ID and + browser_result.get("width") == 1280 and + browser_result.get("height") == 720 and + browser_result.get("automatically_added") is True, + "Tuldok real browser result", browser_result) + emit("tuldok_complete", evidence=str(evidence), + saved_sha256=browser_result.get("saved_sha256")) + except BaseException as error: + failure = error + finally: + if base is not None: + if loaded: + try: + unloaded = rpc(base, "unserve_model", {"request": { + "model_id": MODEL_ID, "provider": "torch", + "profile_id": PROFILE_ID, "model_alias": ALIAS, + }}, timeout=120) + require(unloaded.get("success") and unloaded.get("unloaded"), "FLUX.2 unload", unloaded) + emit("model_unloaded") + except BaseException as error: + failure = failure or error + try: + stopped = rpc(base, "stop_runtime_profile", {"profile_id": PROFILE_ID}, timeout=60) + require(stopped.get("success"), "managed sidecar stop", stopped) + emit("sidecar_stopped") + except BaseException as error: + failure = failure or error + if process.poll() is None: + process.send_signal(signal.SIGINT) + try: + process.wait(timeout=30) + except subprocess.TimeoutExpired: + os.killpg(process.pid, signal.SIGKILL) + process.wait(timeout=10) + failure = failure or RuntimeError("RPC required forced shutdown") + if process.returncode != 0: + failure = failure or RuntimeError(f"RPC exited {process.returncode}; see {log_path}") + emit("rpc_stopped", returncode=process.returncode) + try: + stop_orphaned_sidecars() + except BaseException as error: + failure = failure or error + if guard: + guard.finish() + emit("host_memory_guard", minimum_available_bytes=guard.minimum_available, + low_available_bytes=guard.low_available, tripped=guard.tripped) + if guard.tripped: + failure = RuntimeError(f"host MemAvailable fell below 6 GiB or became unavailable: {guard.low_available}") + events_after = memory_events() + event_delta = {key: value - events_before.get(key, 0) for key, value in events_after.items()} + peak = cgroup_value("memory.peak") + emit("memory", cgroup_peak_bytes=int(peak) if peak and peak != "max" else None, + cgroup_events_after=events_after, cgroup_event_delta=event_delta) + if any(event_delta.get(key, 0) > 0 for key in ("oom", "oom_kill", "oom_group_kill")): + failure = failure or RuntimeError(f"cgroup OOM event occurred: {event_delta}") + if failure: + raise failure + emit("complete", output=str(output), log=str(log_path)) + + +if __name__ == "__main__": + try: + main() + except Exception as error: + emit("failed", error_type=type(error).__name__, error=str(error)) + raise SystemExit(1) from error From a70c0665c66dd951b8e7f8f2f79d70e4b6e01401 Mon Sep 17 00:00:00 2001 From: MrScripty Date: Sun, 27 Sep 2026 03:27:31 -0700 Subject: [PATCH 4/7] Record main AppImage Torch 2.14 FLUX recovery --- .../upstream-v214-cu132-flux2-klein-e2e.md | 13 +++-- .../upstream-version-manager-progress.md | 58 ++++++++++++++----- 2 files changed, 53 insertions(+), 18 deletions(-) diff --git a/docs/plans/torch-diffusion-serving/reports/upstream-v214-cu132-flux2-klein-e2e.md b/docs/plans/torch-diffusion-serving/reports/upstream-v214-cu132-flux2-klein-e2e.md index ac7841cb..148d504d 100644 --- a/docs/plans/torch-diffusion-serving/reports/upstream-v214-cu132-flux2-klein-e2e.md +++ b/docs/plans/torch-diffusion-serving/reports/upstream-v214-cu132-flux2-klein-e2e.md @@ -25,13 +25,16 @@ Five focused Rust admission tests, Rust formatting, a rebuilt debug RPC binary, ## Desktop deployment state -The main selected `v2.14.0` runtime remains the earlier Core (`adapter: none`) -installation; it cannot serve images until replaced through the managed -installer with a FLUX.2-enabled recipe. A new local Linux AppImage and deb were +At the time of this isolated trial, the main selected `v2.14.0` runtime was +still the earlier Core (`adapter: none`) installation. A new local Linux +AppImage and deb were built with the repaired release RPC. The AppImage SHA-256 is `b05e198d0a29ec5049b7afaaa9678afd41bde861e7ce34d5f7e27fabdc60711c`; the deb SHA-256 is `af8d0e8ae81e087806218aa254a1a28a7be656d75a04c0211c22bd7b8d0f17c6`. Both extracted packages matched their staged RPC/resources and passed RPC -`/health` smoke tests. The packaged desktop install/serve/Tuldok flow still -needs acceptance after the main runtime is replaced. Nothing was published. +`/health` smoke tests. The main runtime was subsequently replaced through the +managed installer, and the real AppImage load/gateway/Tuldok flow passed on +2026-09-27 as recorded in the +[version-manager inventory](upstream-version-manager-progress.md). Nothing was +published. diff --git a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md index 7bb8d6ae..46ea708d 100644 --- a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md +++ b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md @@ -1,13 +1,47 @@ # Upstream Torch version manager progress -Date: 2026-09-26 (America/Vancouver). This is an implementation and evidence -inventory. Torch 2.14 FLUX.2 Klein now has isolated real RPC/gateway and Tuldok -browser acceptance; the main selected desktop runtime still has the Core recipe. -The prior +Date: 2026-09-27 (America/Vancouver). This is an implementation and evidence +inventory. Torch 2.14 FLUX.2 Klein now has real AppImage gateway and Tuldok +browser acceptance from the main launcher root. The prior [A1 packaged acceptance](a1-packaged-acceptance.md) applies only to its original -`v2.9.1` / CPython 3.12 / CUDA 13.0 / Linux x86_64 combination. The Tuldok image -evidence below uses a separate, exact Torch 2.10 tuple and does not qualify a -newly resolved upstream release. The isolated Torch 2.14 result is recorded below. +`v2.9.1` / CPython 3.12 / CUDA 13.0 / Linux x86_64 combination. The earlier +Torch 2.10 Tuldok result is separate from the Torch 2.14 results below. + +## 2026-09-27 — Repair the real AppImage FLUX model load + +The user launched the local `Pumas.Library-0.7.0.AppImage` and tried to serve +FLUX.2 Klein. Its Electron log recorded a real sidecar load failure: +`No module named 'diffusers'`. The AppImage bundled the repaired RPC, but the +main selected/default `v2.14.0` installation was the earlier Core recipe with +`adapter: none`. The FLUX-enabled 2.14 installation from the previous trial +was in an isolated launcher root, so it was not available to this AppImage. + +The failed profile, Electron log, Core runtime, selection/default metadata and +profile metadata were preserved under +`launcher-data/cache/torch-qualification/main-v214-flux-repair-20260927/core-backup/`. +With the Torch profile stopped, Pumas selected the existing `torch-runtime-0.1.4` +fallback, cleared the default, removed the inactive 2.14 entry through RPC, +and installed `v2.14.0` again through its managed preview/install flow with +`cu132`, Python 3.14 and `flux2`. The new main-root probe passed Torch +`2.14.0+cu132`, CPU/CUDA operations, sidecar health, and FLUX imports. The +installer log is `launcher-data/logs/install-v2.14.0-1790504224408.log`. +Pumas then selected/defaulted the new 2.14 runtime; its managed startup trial +passed protocol 3 and health. + +The local packaged RPC loaded FLUX Klein from the main root, advertised it in +`/v1/models`, and returned a real 512×512 PNG. Tuldok browser generation, +display, save and automatic collection import passed at 1280×720. The same +checks then passed using the **running AppImage's own** RPC and gateway; its +Tuldok result took 14.7 seconds and saved PNG SHA-256 +`ba32e3506ebee5a9e48c35c506777fa6b946b3ea9d7c45c86d1b6befd454750f`. +At completion, the AppImage remained running with FLUX ready in `/v1/models` +and the `torch-image-acceptance` profile running. The gateway URL observed in +that session was `http://127.0.0.1:41009/v1`; its port is assigned at startup. +Local ignored evidence is in `main-v214-flux-repair-20260927/`, including +`tuldok-appimage/{result.json,saved.png,display.png}` and before/after RPC +snapshots. No new source or release build was needed, and nothing was published. +The successful model load was invoked through the AppImage's RPC; a second +manual click through its model-serving UI was not part of this check. ## 2026-09-26 — Real Torch 2.14 FLUX.2 Klein endpoint and Tuldok acceptance @@ -28,12 +62,10 @@ The full run used a 34 GiB capped scope with no OOM events; 490 memory-limit reclaim events show pressure at the cap. The host stayed at least 20.9 GiB available during that run. The model, profile, RPC and browser were stopped. -This qualifies the isolated v2.14+FLUX tuple and the Pumas endpoint/Tuldok path. -The main selected v2.14 installation is still Core (`adapter: none`) and cannot -serve images. It needs a managed FLUX recipe replacement for desktop use. New -local, unpublished AppImage/deb artifacts include this RPC change and passed -extracted resource and backend-health smokes; packaged desktop install/serve -and Tuldok acceptance remain open. No release was published. +This qualified the isolated v2.14+FLUX tuple and the Pumas endpoint/Tuldok path. +At this 2026-09-26 checkpoint the main selected v2.14 installation was still +Core (`adapter: none`). The 2026-09-27 managed replacement and packaged desktop +acceptance above supersede that deployment gap. No release was published. ## 2026-09-26 — Repair pip rate collection across versions and keep it visible From 937c2b96f51b3c366e1e4d64dc32db269c3ded2e Mon Sep 17 00:00:00 2001 From: MrScripty Date: Sun, 27 Sep 2026 03:48:24 -0700 Subject: [PATCH 5/7] Preserve FLUX image dimensions for Tuldok --- .../upstream-version-manager-progress.md | 54 ++++++++++++++++++- torch-server/diffusion.py | 16 +++++- torch-server/flux2.py | 1 + .../tests_native/test_flux2_dimensions.py | 48 +++++++++++++++++ 4 files changed, 115 insertions(+), 4 deletions(-) create mode 100644 torch-server/tests_native/test_flux2_dimensions.py diff --git a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md index 46ea708d..260c2375 100644 --- a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md +++ b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md @@ -1,12 +1,62 @@ # Upstream Torch version manager progress Date: 2026-09-27 (America/Vancouver). This is an implementation and evidence -inventory. Torch 2.14 FLUX.2 Klein now has real AppImage gateway and Tuldok -browser acceptance from the main launcher root. The prior +inventory. Torch 2.14 FLUX.2 Klein now has real 1920×1080 AppImage gateway and +Tuldok browser acceptance from the main launcher root. The prior [A1 packaged acceptance](a1-packaged-acceptance.md) applies only to its original `v2.9.1` / CPython 3.12 / CUDA 13.0 / Linux x86_64 combination. The earlier Torch 2.10 Tuldok result is separate from the Torch 2.14 results below. +## 2026-09-27 — Repair Tuldok's 1920×1080 generation failure + +After the main-root FLUX installation below, Tuldok requested 1920×1080 from +the AppImage. The sidecar log showed all four GPU inference steps completing, +then `/api/images/generate` returned HTTP 502. Diffusers rounded the requested +1080 height down to 1072 because FLUX requires multiples of 16; the sidecar's +exact-size check correctly rejected that image. Subsequent health/slot checks +passed, so this was a response-size failure rather than a sidecar crash. + +The FLUX adapter now rounds pipeline dimensions **up** to multiples of 16, +then center crops the result to the requested size. Its 1920×1080 request +therefore generates at 1920×1088 and returns an exact 1920×1080 PNG. Other +adapters retain their existing dimension behavior, and both sidecar and gateway +PNG size checks remain in force. The native regression test failed against the +old behavior and passed after the repair; three focused FLUX tests and ten image +request contract tests passed. Independent read-only review found no lifecycle +or security blocker. + +The previous managed FLUX runtime was backed up under +`launcher-data/cache/torch-qualification/main-v214-flux-repair-20260927/pre-dimension-backup/`. +With its profile stopped, Pumas switched to the installed fallback and used its +preview/remove/install flow to materialize the repaired `v2.14.0+cu132` FLUX +sidecar. The new runtime was selected and defaulted. The installer log is +`launcher-data/logs/install-v2.14.0-1790505527380.log`; its source +`diffusion.py` matches the repaired source file. The rebuilt release RPC has +SHA-256 `36af552359f5ef1b522976bbc6cc2c61f0810fbcd69fd5757a2c676ae78e1e92`. +The rebuilt local, unpublished AppImage has SHA-256 +`f12b98fc6d5c2dbfe2b469228ca90d1fdc7ebe09e000e95b6c45461a4885f9a8`. + +The managed gateway loaded and advertised the real FLUX checkpoint, then +returned a 1920×1080 PNG in 23.389 seconds (SHA-256 +`702fc7d018b7d4b933497c3db6bb3c067f0219f137b91d9ea70e69b7ac8a8a19`). +The **rebuilt AppImage's own** RPC loaded the same model. A temporary +1920×1080 variant of Tuldok's real browser acceptance script discovered it +through the AppImage gateway, generated a real image, displayed it at the +requested size, saved it, and automatically imported it into the collection. +The Tuldok job took 23.526 seconds and its saved PNG had the same hash. The +local ignored evidence is under +`main-v214-flux-repair-20260927/tuldok-appimage-1920x1080/`, including +`result.json`, `saved.png`, and `display.png`. The screenshot was visually +inspected. This acceptance used a task-owned Tuldok browser instance and the +AppImage RPC; it did not click the AppImage's model-serving UI again. The +observed gateway was `http://127.0.0.1:41253/v1`; its port changes on restart. +The user's separately running Tuldok instance on port 8091 then submitted its +own 1920×1080 FLUX request to that same gateway. Job +`c06969191f5542e09c84931737fbc661` completed with one image and no error; +`/api/samples` contained `synthetic-c0696919-00001.png` at 1920×1080 in +session `2026-09-12-desk`. The user also confirmed image generation worked. +No release was published. + ## 2026-09-27 — Repair the real AppImage FLUX model load The user launched the local `Pumas.Library-0.7.0.AppImage` and tried to serve diff --git a/torch-server/diffusion.py b/torch-server/diffusion.py index 26881b73..eff638a7 100644 --- a/torch-server/diffusion.py +++ b/torch-server/diffusion.py @@ -18,6 +18,8 @@ class GenerationCancelled(RuntimeError): class DiffusionPipelineAdapter: """Shared denoising cancellation and offload cleanup ownership.""" + dimension_multiple = 1 + def generate(self, prompt: str, width: int, height: int, seed: int, cancel: threading.Event): def checkpoint(_pipeline, _step, _timestep, callback_kwargs): if cancel.is_set(): @@ -26,12 +28,15 @@ def checkpoint(_pipeline, _step, _timestep, callback_kwargs): if cancel.is_set(): raise GenerationCancelled("Generation cancelled") + multiple = self.dimension_multiple + pipeline_width = (width + multiple - 1) // multiple * multiple + pipeline_height = (height + multiple - 1) // multiple * multiple try: with torch.inference_mode(): result = self.pipeline( prompt=prompt, - width=width, - height=height, + width=pipeline_width, + height=pipeline_height, num_inference_steps=self.steps, guidance_scale=self.guidance, generator=torch.Generator(device="cpu").manual_seed(seed), @@ -39,6 +44,13 @@ def checkpoint(_pipeline, _step, _timestep, callback_kwargs): ).images[0] if cancel.is_set(): raise GenerationCancelled("Generation cancelled") + if (pipeline_width, pipeline_height) != (width, height) and result.size == ( + pipeline_width, + pipeline_height, + ): + left = (pipeline_width - width) // 2 + top = (pipeline_height - height) // 2 + result = result.crop((left, top, left + width, top + height)) return result finally: # CUDA kernels and offload hooks must finish before the caller releases diff --git a/torch-server/flux2.py b/torch-server/flux2.py index 24c90e01..68e75172 100644 --- a/torch-server/flux2.py +++ b/torch-server/flux2.py @@ -60,6 +60,7 @@ class Flux2Klein(DiffusionPipelineAdapter): steps = 4 guidance = 1.0 memory_policy = "scaled_fp8_to_bf16_sequential_cpu_offload" + dimension_multiple = 16 def __init__(self, checkpoint, encoder_path, vae_path, device): from diffusers import ( diff --git a/torch-server/tests_native/test_flux2_dimensions.py b/torch-server/tests_native/test_flux2_dimensions.py new file mode 100644 index 00000000..46eb14d8 --- /dev/null +++ b/torch-server/tests_native/test_flux2_dimensions.py @@ -0,0 +1,48 @@ +"""FLUX.2 preserves requested image dimensions across Diffusers rounding.""" + +from pathlib import Path +import sys +import threading +from types import SimpleNamespace +import unittest +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from PIL import Image # noqa: E402 +from flux2 import Flux2Klein # noqa: E402 + + +class Flux2DimensionTests(unittest.TestCase): + def test_1920x1080_rounds_inference_up_then_center_crops_for_provider(self): + seen = {} + + class Pipeline: + def __call__(self, **kwargs): + seen.update(kwargs) + # Match Diffusers' current behavior when it receives a size + # that is not divisible by 16. + width = kwargs["width"] // 16 * 16 + height = kwargs["height"] // 16 * 16 + image = Image.new("RGB", (width, height)) + image.putpixel((0, 4), (255, 0, 0)) + image.putpixel((0, height - 5), (0, 0, 255)) + return SimpleNamespace(images=[image]) + + def maybe_free_model_hooks(self): + pass + + adapter = Flux2Klein.__new__(Flux2Klein) + adapter.pipeline = Pipeline() + + with patch("torch.cuda.synchronize"): + image = adapter.generate("test", 1920, 1080, 7, threading.Event()) + + self.assertEqual((seen["width"], seen["height"]), (1920, 1088)) + self.assertEqual(image.size, (1920, 1080)) + self.assertEqual(image.getpixel((0, 0)), (255, 0, 0)) + self.assertEqual(image.getpixel((0, 1079)), (0, 0, 255)) + + +if __name__ == "__main__": + unittest.main() From 04e7f1568f00693c0ef26c77e0150e5e4dd112ea Mon Sep 17 00:00:00 2001 From: MrScripty Date: Sun, 27 Sep 2026 04:23:26 -0700 Subject: [PATCH 6/7] Support Nunchaku image sizes in bundled Torch runtime --- .../upstream-version-manager-progress.md | 64 +++++++++++++++++++ torch-server/diffusion.py | 1 + .../tests_native/test_nunchaku_dimensions.py | 47 ++++++++++++++ 3 files changed, 112 insertions(+) create mode 100644 torch-server/tests_native/test_nunchaku_dimensions.py diff --git a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md index 260c2375..3b927d2b 100644 --- a/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md +++ b/docs/plans/torch-diffusion-serving/reports/upstream-version-manager-progress.md @@ -7,6 +7,70 @@ Tuldok browser acceptance from the main launcher root. The prior `v2.9.1` / CPython 3.12 / CUDA 13.0 / Linux x86_64 combination. The earlier Torch 2.10 Tuldok result is separate from the Torch 2.14 results below. +## 2026-09-27 — Recover Nunchaku serving after FLUX + +After unloading FLUX, the user tried to serve +`diffusion/nunchaku-ai/nunchaku-z-image-turbo` from the selected main-root +`v2.14.0+cu132` FLUX-only runtime. Its sidecar log recorded +`ModuleNotFoundError: No module named 'nunchaku'` before checkpoint loading. +The selected runtime's probe had marked Nunchaku `not selected`; generic +`image_generation` protocol capability alone did not mean every adapter was +installed. The selected Nunchaku wheel is pinned by +`torch-server/resolve_runtime.py` and the bundled lock to Nunchaku +`1.2.0+torch2.9`, CPython 3.12, Torch 2.9.1+cu130, and Linux x86_64. Commit +`04e4ce2b7` introduced that exact URL/hash and compatibility check on +2026-09-23. The original CUDA 12.8 candidate failed a real native import +because it required `libcudart.so.13`; the corrected CUDA 13.0 recipe is +recorded in [the runtime report](runtime.md). This is the currently qualified +binary combination, not a permanent Nunchaku restriction. The official +[Nunchaku releases](https://github.com/nunchux-ai/nunchaku/releases) include +newer Torch-targeted wheels, but no Torch 2.14/Python 3.14 wheel was qualified +for this Pumas runtime. + +Through the rebuilt AppImage's managed RPC, Pumas installed the pinned +`v2.9.1` `bundled` runtime alongside the preserved `v2.14.0` install. Its +Torch/CUDA and both adapter import probes passed; real model inference remained +unverified until the tests below. A managed protocol-3 startup trial and real +Nunchaku model load passed. A 1280×720 gateway image also passed. The next real +1920×1080 request failed before inference because the installed Z-Image +pipeline requires dimensions divisible by 16. `NunchakuZImage` now opts into +the shared adapter's round-up and center-crop behavior: infer at 1920×1088, +return exactly 1920×1080. A native regression test reproduced the old failure +and passed with the fix; the existing FLUX dimension test still passed. +Independent read-only review found no lifecycle or output-validation blocker. + +The pre-fix 2.9.1 environment was preserved under +`launcher-data/cache/torch-qualification/main-v214-nunchaku-repair-20260927/pre-dimension-backup/`. +Pumas removed the inactive version and reinstalled it with the rebuilt embedded +sidecar through a fresh managed preview. The second installer log is +`launcher-data/logs/install-v2.9.1-1790507472804.log`; its materialized +`diffusion.py` matches current source. The rebuilt release RPC SHA-256 is +`4f50eb364c895021bc127803cfd9e7fca89e8059d07c35911e8452eefe5efe64`. +The local, unpublished AppImage SHA-256 is +`d8caa386a95d904e349c4dcaaf8bf54432fc9b842633ae3c7857ff4c688053d8`; +the AppImage and deb passed the Linux artifact checker. + +The repaired managed gateway loaded Nunchaku and returned a real 1920×1080 +PNG in 20.081 seconds, SHA-256 +`513295f98937498f028a1bff8af879f578347b410a99fe27538b95befa16fbfe`. +The **rebuilt AppImage's own** RPC then loaded Nunchaku. A task-owned Tuldok +browser discovered it, generated a real 1920×1080 image, displayed it, saved +it, and automatically imported it into its temporary collection in 20.751 +seconds; saved PNG SHA-256 +`a70255be804e829430f10e244deed53f004bed7edf9ac18a4081b7ef6fc71fdc`. +The displayed image was visually inspected. Ignored evidence is under +`main-v214-nunchaku-repair-20260927/tuldok-appimage-1920x1080/`. The gateway +was `http://127.0.0.1:44581/v1` in this session; its port changes on restart. +The model load used the AppImage RPC, not a repeat click through its UI. + +At this checkpoint `v2.9.1` is the active Torch version for Nunchaku, while +`v2.14.0` remains installed and the configured default for FLUX. Torch version +selection is global: the two versions cannot serve models simultaneously, and +changing the active version requires unloading models and stopping the Torch +profile. The current managed 2.9.1 bundled runtime passed both adapter import +probes, but real FLUX generation in **that exact new runtime** was not rerun +here. No release was published. + ## 2026-09-27 — Repair Tuldok's 1920×1080 generation failure After the main-root FLUX installation below, Tuldok requested 1920×1080 from diff --git a/torch-server/diffusion.py b/torch-server/diffusion.py index eff638a7..8566fb73 100644 --- a/torch-server/diffusion.py +++ b/torch-server/diffusion.py @@ -65,6 +65,7 @@ class NunchakuZImage(DiffusionPipelineAdapter): steps = 8 guidance = 0.0 memory_policy = "sequential_cpu_offload" + dimension_multiple = 16 def __init__(self, checkpoint: str, pipeline_path: str, device: torch.device): from diffusers import ZImagePipeline diff --git a/torch-server/tests_native/test_nunchaku_dimensions.py b/torch-server/tests_native/test_nunchaku_dimensions.py new file mode 100644 index 00000000..e2d352ac --- /dev/null +++ b/torch-server/tests_native/test_nunchaku_dimensions.py @@ -0,0 +1,47 @@ +"""Nunchaku Z-Image preserves requested dimensions across pipeline alignment.""" + +from pathlib import Path +import sys +import threading +from types import SimpleNamespace +import unittest +from unittest.mock import patch + +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +from PIL import Image # noqa: E402 +from diffusion import NunchakuZImage # noqa: E402 + + +class NunchakuDimensionTests(unittest.TestCase): + def test_1920x1080_rounds_inference_up_then_center_crops(self): + seen = {} + + class Pipeline: + def __call__(self, **kwargs): + width, height = kwargs["width"], kwargs["height"] + if width % 16 or height % 16: + raise ValueError("ZImagePipeline requires dimensions divisible by 16") + seen.update(kwargs) + image = Image.new("RGB", (width, height)) + image.putpixel((0, 4), (255, 0, 0)) + image.putpixel((0, height - 5), (0, 0, 255)) + return SimpleNamespace(images=[image]) + + def maybe_free_model_hooks(self): + pass + + adapter = NunchakuZImage.__new__(NunchakuZImage) + adapter.pipeline = Pipeline() + + with patch("torch.cuda.synchronize"): + image = adapter.generate("test", 1920, 1080, 7, threading.Event()) + + self.assertEqual((seen["width"], seen["height"]), (1920, 1088)) + self.assertEqual(image.size, (1920, 1080)) + self.assertEqual(image.getpixel((0, 0)), (255, 0, 0)) + self.assertEqual(image.getpixel((0, 1079)), (0, 0, 255)) + + +if __name__ == "__main__": + unittest.main() From 1a29717d9bbfa441911611fbf7e50de12a88c472 Mon Sep 17 00:00:00 2001 From: MrScripty Date: Tue, 29 Sep 2026 07:14:26 -0700 Subject: [PATCH 7/7] test(torch): wait for install choices before selecting build --- frontend/src/components/TorchInstallPreview.test.tsx | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/frontend/src/components/TorchInstallPreview.test.tsx b/frontend/src/components/TorchInstallPreview.test.tsx index c81b8e91..11e593f6 100644 --- a/frontend/src/components/TorchInstallPreview.test.tsx +++ b/frontend/src/components/TorchInstallPreview.test.tsx @@ -186,10 +186,17 @@ describe('TorchInstallPreview', () => { }); it('falls back to automatic build when the quick options omit defaultBuild', async () => { - getOptions.mockResolvedValue({ ...options, defaultBuild: undefined }); + let resolveOptions!: (value: TorchRuntimeOptions) => void; + getOptions.mockImplementation(() => new Promise((resolve) => { + resolveOptions = resolve; + })); const onInstall = vi.fn(); render(); - fireEvent.click(await screen.findByRole('button', { name: 'Install Torch' })); + expect(screen.getByRole('button', { name: 'Install Torch' })).toBeDisabled(); + await act(async () => { resolveOptions({ ...options, defaultBuild: undefined }); }); + const installButton = screen.getByRole('button', { name: 'Install Torch' }); + expect(installButton).toBeEnabled(); + fireEvent.click(installButton); await waitFor(() => expect(onInstall).toHaveBeenCalledWith('selection-token')); expect(getSelection).toHaveBeenCalledWith({ tag: 'v2.14.0', build: 'auto', python: 'auto', adapter: 'none' }); });