Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
33 changes: 31 additions & 2 deletions .github/workflows/ci-tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,9 @@ name: CI testing

on:
push:
branches: [main]
pull_request:
branches: [main]

permissions:
contents: read
Expand All @@ -15,7 +17,7 @@ jobs:
uses: ./.github/workflows/build-package.yml


test:
tests-cpu:
name: Pytest
runs-on: ${{ matrix.os }}
timeout-minutes: 30
Expand All @@ -41,5 +43,32 @@ jobs:
PYTHONHASHSEED: '0'
PYTHONPATH: .
run: |
uv run --no-sync python -m coverage run --branch --source=cutie -m pytest tests -q
uv run --no-sync python -m coverage run --branch --source=cutie -m pytest tests
uv run --no-sync python -m coverage report --show-missing


tests-gpu:
name: Pytest (GPU)
runs-on: Roboflow-GPU-VM-Runner
timeout-minutes: 30
env:
UV_TORCH_BACKEND: auto
steps:
- name: Print GPU information
run: nvidia-smi
- name: Check out source
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Set up UV and Python
uses: astral-sh/setup-uv@c771a70e6277c0a99b617c7a806ffedaca235ff9 # v9.0.0
with:
version: '0.9.26'
python-version: '3.12'
activate-environment: true
- name: Install all extras and test dependency group
run: uv pip install --group tests --strict '.[inference,evaluation,train,gui,video,data]'
- name: Run GPU-accelerated device-parity tests
env:
PYTEST_DISABLE_PLUGIN_AUTOLOAD: '1'
PYTHONHASHSEED: '0'
PYTHONPATH: .
run: uv run --no-sync pytest tests
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -148,6 +148,7 @@ dmypy.json
.idea/

# AI assets
.developments/
.reports/

# env files
Expand Down
7 changes: 2 additions & 5 deletions cutie/inference/memory_manager.py
Original file line number Diff line number Diff line change
Expand Up @@ -244,11 +244,8 @@ def add_memory(
for obj_id, obj in enumerate(objects):
if obj in self.obj_v:
# Keep embedding sums and counts for the object transformer's streaming average.
last_acc = self.obj_v[obj][:, :, -1]
new_acc = last_acc + obj_value[:, obj_id, :, -1]

self.obj_v[obj][:, :, :-1] = self.obj_v[obj][:, :, :-1] + obj_value[:, obj_id, :, :-1]
self.obj_v[obj][:, :, -1] = new_acc
self.obj_v[obj][:, :, :-1].add_(obj_value[:, obj_id, :, :-1])
self.obj_v[obj][:, :, -1].add_(obj_value[:, obj_id, :, -1])
else:
self.obj_v[obj] = obj_value[:, obj_id]

Expand Down
2 changes: 1 addition & 1 deletion cutie/model/big_modules.py
Original file line number Diff line number Diff line change
Expand Up @@ -80,7 +80,7 @@ def __init__(self, model_cfg: DictConfig):

def forward(self, x: torch.Tensor, *, need_s: bool, need_e: bool) -> (torch.Tensor, torch.Tensor, torch.Tensor):
x = self.pix_feat_proj(x)
shrinkage = self.d_proj(x) ** 2 + 1 if (need_s) else None
shrinkage = self.d_proj(x).pow(2).add_(1) if (need_s) else None
selection = torch.sigmoid(self.e_proj(x)) if (need_e) else None

return self.key_proj(x), shrinkage, selection
Expand Down
2 changes: 1 addition & 1 deletion cutie/model/channel_attn.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,4 +31,4 @@ def forward(self, x: torch.Tensor) -> torch.Tensor:
w = self.pool(x).view(b, 1, c)
w = self.conv(w).transpose(-1, -2).unsqueeze(-1).sigmoid() # B*C*1*1

return x * w + self.downsample(r) if self.residual else x * w
return torch.addcmul(self.downsample(r), x, w) if self.residual else x * w
5 changes: 3 additions & 2 deletions cutie/model/cutie.py
Original file line number Diff line number Diff line change
Expand Up @@ -54,7 +54,8 @@ def _get_others(self, masks: torch.Tensor) -> torch.Tensor:
return (masks.sum(dim=1, keepdim=True) - masks).clamp(0, 1) if num_objects >= 1 else torch.zeros_like(masks)

def encode_image(self, image: torch.Tensor) -> (Iterable[torch.Tensor], torch.Tensor):
image = (image - self.pixel_mean) / self.pixel_std
# sub() copies (never mutates the caller's frame); only the fresh copy is div_'d in place
image = image.sub(self.pixel_mean).div_(self.pixel_std)
ms_image_feat = self.pixel_encoder(image)
return ms_image_feat, self.pix_feat_proj(ms_image_feat[0])

Expand All @@ -69,7 +70,7 @@ def encode_mask(
chunk_size: int = -1,
need_weights: bool = False,
) -> (torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor):
image = (image - self.pixel_mean) / self.pixel_std
image = image.sub(self.pixel_mean).div_(self.pixel_std)
others = self._get_others(masks)
mask_value, new_sensory = self.mask_encoder(
image,
Expand Down
2 changes: 1 addition & 1 deletion cutie/model/group_modules.py
Original file line number Diff line number Diff line change
Expand Up @@ -89,7 +89,7 @@ def forward(self, x: torch.Tensor, g: torch.Tensor, skip_expand: bool = False) -
elif self.method == 'mulcat':
g = torch.cat([x * g, g], dim=2)
elif self.method == 'muladd':
g = x * g + g
g = torch.addcmul(g, x, g)
else:
raise NotImplementedError

Expand Down
17 changes: 11 additions & 6 deletions cutie/model/utils/memory_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -30,16 +30,18 @@ def get_similarity(
# See XMem's appendix for derivation
mk = mk.transpose(1, 2)
a_sq = mk.pow(2) @ qe
two_ab = 2 * (mk @ (qk * qe))
two_ab = (mk @ (qk * qe)).mul_(2)
b_sq = (qe * qk.pow(2)).sum(1, keepdim=True)
similarity = -a_sq + two_ab - b_sq
similarity = two_ab.sub_(a_sq).sub_(b_sq)
else:
# similar to STCN if we don't have the selection term
a_sq = mk.pow(2).sum(1).unsqueeze(2)
two_ab = 2 * (mk.transpose(1, 2) @ qk)
similarity = -a_sq + two_ab
two_ab = (mk.transpose(1, 2) @ qk).mul_(2)
similarity = two_ab.sub_(a_sq)

return similarity * ms / math.sqrt(CK) if ms is not None else similarity / math.sqrt(CK) # B*N*HW
if ms is not None:
return similarity.mul_(ms).div_(math.sqrt(CK))
return similarity.div_(math.sqrt(CK)) # B*N*HW


def do_softmax(
Expand All @@ -63,7 +65,10 @@ def do_softmax(
affinity = torch.zeros_like(similarity).scatter_(1, indices, x_exp) # B*N*HW
else:
maxes = torch.max(similarity, dim=1, keepdim=True)[0]
x_exp = torch.exp(similarity - maxes)
# exp_() is safe (sub() output is a fresh throwaway tensor); div_() is NOT —
# Exp's backward needs its own output value preserved, so the final
# normalization must stay out-of-place (confirmed by backward-parity test).
x_exp = similarity.sub(maxes).exp_()
x_exp_sum = torch.sum(x_exp, dim=1, keepdim=True)
affinity = x_exp / x_exp_sum
indices = None
Expand Down
6 changes: 5 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,11 @@ quote-style = "single"
ignore-words-list = ["MOSE", "concurent", "indx"]

[tool.pytest.ini_options]
testpaths = ["tests"]
testpaths = ["cutie", "tests"]
addopts = [
"--color=yes",
"--doctest-modules",
]

[tool.ruff.lint]
# Ruff-only baseline. `PL` enables Ruff's Pylint-origin rules; explicit codes
Expand Down
85 changes: 85 additions & 0 deletions scripts/bench_vram_mps.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,85 @@
"""Standalone MPS VRAM bench for the memory-read hot path.

Not a pytest assertion — MPS allocator memory numbers are noisy run-to-run, so
this is evidence to eyeball (before/after the in-place-op refactor), not a CI
gate. Correctness is covered by tests/test_mps_parity.py; this script only
measures peak Apple-GPU memory and wall-clock for cutie.model.utils.memory_utils
get_similarity + do_softmax at a realistic memory-bank size.

Usage:
python scripts/bench_vram_mps.py [--frames N] [--hw N] [--ck N] [--iters N]
"""

import argparse
import time

import torch
from torch import mps

from cutie.model.utils.memory_utils import do_softmax, get_similarity


def _build_inputs(
*, batch: int, ck: int, num_memory_frames: int, hw: int, device: str
) -> tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]:
n = num_memory_frames * hw
mk = torch.randn(batch, ck, n, device=device)
ms = torch.rand(batch, 1, n, device=device)
qk = torch.randn(batch, ck, hw, device=device)
qe = torch.rand(batch, ck, hw, device=device)
return mk, ms, qk, qe


def _run_once(mk: torch.Tensor, ms: torch.Tensor, qk: torch.Tensor, qe: torch.Tensor) -> torch.Tensor:
similarity = get_similarity(mk, ms, qk, qe)
return do_softmax(similarity)


def bench(*, batch: int, ck: int, num_memory_frames: int, hw: int, iters: int) -> None:
if iters < 1:
raise ValueError('iters must be >= 1')
if not torch.backends.mps.is_available():
print('MPS not available on this machine — nothing to bench.')
return
Comment thread
Copilot marked this conversation as resolved.

device = 'mps'
mk, ms, qk, qe = _build_inputs(batch=batch, ck=ck, num_memory_frames=num_memory_frames, hw=hw, device=device)

# warm up (first MPS dispatch pays kernel-compile cost, not representative)
_run_once(mk, ms, qk, qe)
torch.mps.synchronize()

mps.empty_cache()
baseline_allocated = mps.current_allocated_memory()

start = time.perf_counter()
for _ in range(iters):
affinity = _run_once(mk, ms, qk, qe)
torch.mps.synchronize()
elapsed = time.perf_counter() - start

peak_allocated = mps.driver_allocated_memory()

print('MPS VRAM bench — cutie.model.utils.memory_utils (get_similarity + do_softmax)')
print(f' shapes: batch={batch} ck={ck} memory_frames={num_memory_frames} hw={hw} -> N={num_memory_frames * hw}')
print(f' iters: {iters}')
print(f' baseline allocated (post-warmup, pre-loop): {baseline_allocated / 2**20:.2f} MiB')
print(f' driver allocated (peak, post-loop): {peak_allocated / 2**20:.2f} MiB')
print(f' wall-clock: {elapsed:.4f}s total, {elapsed / iters * 1000:.3f}ms/iter')
print(f' output shape: {tuple(affinity.shape)}')


def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument('--batch', type=int, default=1)
parser.add_argument('--ck', type=int, default=64, help='key channel dim')
parser.add_argument('--frames', type=int, default=20, dest='num_memory_frames', help='accumulated memory frames')
parser.add_argument('--hw', type=int, default=30 * 54, help='flattened spatial size (H*W/patch)')
parser.add_argument('--iters', type=int, default=50)
args = parser.parse_args()

bench(batch=args.batch, ck=args.ck, num_memory_frames=args.num_memory_frames, hw=args.hw, iters=args.iters)


if __name__ == '__main__':
main()
129 changes: 129 additions & 0 deletions tests/test_device_parity.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,129 @@
"""CPU-vs-accelerator parity guardrails for pure-math ops slated for VRAM optimization.

These tests prove `cutie/model/utils/memory_utils.py` and
`cutie/model/channel_attn.py` produce numerically consistent results on every
GPU backend PyTorch supports on this codebase's target platforms - CUDA and
Apple's MPS - before any device-related refactor touches them. Every existing
test in this suite otherwise runs CPU-only; this file is the first to actually
execute tensor ops on an accelerator, so each backend's tests are guarded by a
real `is_available()` check rather than being unconditionally skipped. A
machine with only one backend (e.g. this MacBook has MPS, no CUDA) still runs
that backend's cases for real and simply skips the other - both are exercised
wherever hardware allows, neither is assumed absent.
"""

import copy

import pytest
import torch

from cutie.model.channel_attn import CAResBlock
from cutie.model.utils.memory_utils import do_softmax, get_similarity

# Backends to check parity against CPU, each independently skipped when its
# hardware isn't present - never assume one backend stands in for the other.
_ACCELERATOR_DEVICES = [
pytest.param(
'cuda',
marks=pytest.mark.skipif(not torch.cuda.is_available(), reason='CUDA not available on this machine'),
),
pytest.param(
'mps',
marks=pytest.mark.skipif(not torch.backends.mps.is_available(), reason='MPS not available on this machine'),
),
]

# Both CUDA and MPS accumulate in float32 and may reduce in a different order
# than CPU BLAS, so results are close but not bit-identical. These tolerances
# were determined empirically against this repo's ops (see module docstring);
# widen only with a concrete numerical justification.
_ATOL = 1e-5
_RTOL = 1e-4


def _build_similarity_inputs() -> tuple[torch.Tensor, torch.Tensor, torch.Tensor, torch.Tensor]:
"""Build small mk/ms/qk/qe tensors on CPU via the seeded RNG."""
mk = torch.randn(2, 4, 3)
ms = torch.rand(2, 1, 3) + 0.5
qk = torch.randn(2, 4, 5)
qe = torch.rand(2, 4, 5) + 0.5
return mk, ms, qk, qe


class TestGetSimilarityDeviceParity:
"""Guardrail: `get_similarity` must agree between CPU and each accelerator backend."""

@pytest.mark.parametrize('device', _ACCELERATOR_DEVICES)
@pytest.mark.parametrize(
'use_qe',
[
pytest.param(True, id='qe_present'),
pytest.param(False, id='qe_none'),
],
)
def test_get_similarity_matches_between_cpu_and_device(self, device: str, use_qe: bool) -> None:
"""Same CPU-built tensors moved to the accelerator give results close to the CPU run."""
mk_cpu, ms_cpu, qk_cpu, qe_cpu = _build_similarity_inputs()
qe_cpu = qe_cpu if use_qe else None

mk_dev, ms_dev, qk_dev = mk_cpu.to(device), ms_cpu.to(device), qk_cpu.to(device)
qe_dev = qe_cpu.to(device) if qe_cpu is not None else None

cpu_result = get_similarity(mk_cpu, ms_cpu, qk_cpu, qe_cpu)
device_result = get_similarity(mk_dev, ms_dev, qk_dev, qe_dev)

torch.testing.assert_close(cpu_result, device_result.cpu(), atol=_ATOL, rtol=_RTOL)


class TestDoSoftmaxDeviceParity:
"""Guardrail: `do_softmax` must agree between CPU and each accelerator backend."""

@pytest.mark.parametrize('device', _ACCELERATOR_DEVICES)
@pytest.mark.parametrize(
'top_k',
[
pytest.param(None, id='dense_softmax'),
pytest.param(2, id='top_k_subset'),
],
)
def test_do_softmax_matches_between_cpu_and_device(self, device: str, top_k: int | None) -> None:
"""Same CPU-built similarity tensor moved to the accelerator gives a close affinity map."""
similarity_cpu = torch.randn(2, 5, 3)
similarity_dev = similarity_cpu.to(device)

cpu_result = do_softmax(similarity_cpu, top_k=top_k, inplace=False)
device_result = do_softmax(similarity_dev, top_k=top_k, inplace=False)

torch.testing.assert_close(cpu_result, device_result.cpu(), atol=_ATOL, rtol=_RTOL)


class TestCAResBlockDeviceParity:
"""Guardrail: `CAResBlock` forward pass must agree between CPU and each accelerator backend."""

@pytest.mark.parametrize('device', _ACCELERATOR_DEVICES)
def test_ca_res_block_forward_matches_between_cpu_and_device(self, device: str) -> None:
"""Identical weights on CPU vs the accelerator produce a forward output within tolerance."""
module_cpu = CAResBlock(4, 4).eval()
# deepcopy before moving preserves exact weights - two independently
# constructed modules would diverge even under the seed fixture, since
# Conv2d init happens at construction time.
module_dev = copy.deepcopy(module_cpu).to(device).eval()
x_cpu = torch.randn(1, 4, 8, 8)
x_dev = x_cpu.to(device)

with torch.no_grad():
cpu_result = module_cpu(x_cpu)
device_result = module_dev(x_dev)

torch.testing.assert_close(cpu_result, device_result.cpu(), atol=_ATOL, rtol=_RTOL)


@pytest.mark.parametrize('device', _ACCELERATOR_DEVICES)
def test_get_similarity_actually_executes_on_device(device: str) -> None:
"""Result tensor stays on the accelerator - proves no silent CPU fallback occurred."""
mk_cpu, ms_cpu, qk_cpu, qe_cpu = _build_similarity_inputs()
mk, ms, qk, qe = (t.to(device) for t in (mk_cpu, ms_cpu, qk_cpu, qe_cpu))

result = get_similarity(mk, ms, qk, qe)

assert result.device.type == device
Loading
Loading