Browse Source

Keep queued jobs when a printer refuses the upload (#3210)

maziggy 2 days ago
parent
commit
9a5ab06ed7

+ 8 - 1
backend/app/services/bambu_ftp.py

@@ -1275,7 +1275,14 @@ class BambuFTPClient:
             return False
             return False
         except (OSError, ftplib.Error) as e:
         except (OSError, ftplib.Error) as e:
             logger.error("FTP upload failed for %s: %s (type: %s)", remote_path, e, type(e).__name__)
             logger.error("FTP upload failed for %s: %s (type: %s)", remote_path, e, type(e).__name__)
-            self.last_failure = FtpFailure(FtpFailureKind.NETWORK, str(e), _ftp_reply_code(e))
+            code = _ftp_reply_code(e)
+            # 452 is "insufficient storage" -- a 4xx, so ftplib raises it as
+            # error_temp rather than error_perm, but it is the printer talking
+            # about its card just as 552/553 are. As NETWORK it would read as a
+            # connection problem, and the queue would retry it forever (#3210).
+            is_storage = isinstance(e, ftplib.Error) and code == "452"
+            kind = FtpFailureKind.STORAGE if is_storage else FtpFailureKind.NETWORK
+            self.last_failure = FtpFailure(kind, str(e), code)
             return False
             return False
 
 
     def upload_bytes(self, data: bytes, remote_path: str) -> bool:
     def upload_bytes(self, data: bytes, remote_path: str) -> bool:

+ 138 - 1
backend/app/services/print_scheduler.py

@@ -32,6 +32,7 @@ from backend.app.models.spool_assignment import SpoolAssignment
 from backend.app.models.spoolman_slot_assignment import SpoolmanSlotAssignment
 from backend.app.models.spoolman_slot_assignment import SpoolmanSlotAssignment
 from backend.app.services import drying_preflight, print_dispatch_context, stock_forecast
 from backend.app.services import drying_preflight, print_dispatch_context, stock_forecast
 from backend.app.services.bambu_ftp import (
 from backend.app.services.bambu_ftp import (
+    FtpFailureKind,
     FtpFailureReport,
     FtpFailureReport,
     UploadCancelled,
     UploadCancelled,
     cache_3mf_download,
     cache_3mf_download,
@@ -350,6 +351,35 @@ _ACTIVE_PRINT_STATES: frozenset[str] = frozenset({"PREPARE", "SLICING", "RUNNING
 # force-reconnect on the very next attempt — while still bounding the loop.
 # force-reconnect on the very next attempt — while still bounding the loop.
 DISPATCH_MAX_ATTEMPTS = 3
 DISPATCH_MAX_ATTEMPTS = 3
 
 
+# Upload failures that mean the file never reached the printer's storage: the
+# file service refused or never answered (#3210). These put the item back in
+# the queue instead of failing it, because nothing about the job is wrong --
+# and failing it left the printer idle, so the next pass handed it the next
+# item, which failed the same way. One P2S whose file service was out of
+# connection slots ate 43 queued jobs in ten minutes that way.
+#
+# Not here: AUTH (a wrong access code needs the user), STORAGE (a full or
+# missing card does too), NOT_FOUND (a Bambuddy-side path problem), UNKNOWN,
+# and an upload that overran its deadline -- each of those would fail the same
+# way on every retry.
+_UPLOAD_REQUEUE_KINDS: frozenset[FtpFailureKind] = frozenset(
+    {FtpFailureKind.HANDSHAKE, FtpFailureKind.COOLOFF, FtpFailureKind.TIMEOUT, FtpFailureKind.NETWORK}
+)
+
+# How long a printer whose upload was put back stays out of dispatch (#3210).
+# The first window matches the FTP client's own handshake cool-off, so the
+# retry lands after the client would talk to the printer again anyway. Each
+# further refusal in a row doubles it, up to the cap: #3210's printer refused
+# for some 40 hours, and every retry costs a preheat cycle where preheat is on,
+# five connection attempts and a page of log. A successful upload resets it.
+UPLOAD_FAILURE_BACKOFF_SECONDS = 300
+UPLOAD_FAILURE_BACKOFF_MAX_SECONDS = 3600
+
+
+def _upload_backoff_seconds(refusals: int) -> int:
+    """Backoff after the *refusals*-th refused upload in a row (1-based)."""
+    return min(UPLOAD_FAILURE_BACKOFF_SECONDS * 2 ** max(refusals - 1, 0), UPLOAD_FAILURE_BACKOFF_MAX_SECONDS)
+
 
 
 @dataclass(slots=True)
 @dataclass(slots=True)
 class _ModelCandidate:
 class _ModelCandidate:
@@ -1053,6 +1083,20 @@ class PrintScheduler:
         # moment it connects.
         # moment it connects.
         self._wake_failures: dict[int, float] = {}
         self._wake_failures: dict[int, float] = {}
         self._wake_failure_cooloff = 600  # seconds
         self._wake_failure_cooloff = 600  # seconds
+        # Printers whose last upload never reached them, mapped to the monotonic
+        # time they may be dispatched to again (#3210). Same expire-on-read shape
+        # as `_wake_failures`. Without it, a printer that cannot take files is
+        # idle on every pass, so it is picked on every pass.
+        self._upload_backoff: dict[int, float] = {}
+        # Refused uploads in a row per printer, which sets the next backoff
+        # window. Cleared by a successful upload to that printer.
+        self._upload_refusals: dict[int, int] = {}
+        # Printers whose current run of refusals has already sent its one
+        # "job waiting" notification. Each retry clears the item's waiting
+        # reason and the next refusal sets it again, which `hold_item` reads as
+        # a new reason; without this, every retry would notify. Cleared with
+        # `_upload_refusals`.
+        self._upload_refusal_notified: set[int] = set()
         # Track which printers are currently auto-drying (printer_id -> start timestamp)
         # Track which printers are currently auto-drying (printer_id -> start timestamp)
         self._drying_in_progress: dict[int, float] = {}
         self._drying_in_progress: dict[int, float] = {}
         # Per-AMS memory of the auto-drying cycles WE armed, keyed by
         # Per-AMS memory of the auto-drying cycles WE armed, keyed by
@@ -1626,6 +1670,34 @@ class PrintScheduler:
             # generalises that workaround instead of repeating it per case.
             # generalises that workaround instead of repeating it per case.
             dispatching_printers: set[int] = set(busy_printers)
             dispatching_printers: set[int] = set(busy_printers)
 
 
+            # Printers whose last upload never reached them (#3210). After the
+            # snapshot above on purpose: such a printer is idle, not about to
+            # print, and auto-drying must keep treating it as idle.
+            now_mono = time.monotonic()
+            # In backoff this pass. Kept apart from busy_printers so keep-warm
+            # can leave them out: there is no print to keep a bed warm for.
+            backoff_printers: set[int] = set()
+            # In backoff, and the "job waiting" notification for this run of
+            # refusals has gone out already. Their holds are written silently.
+            silent_hold_printers: set[int] = set()
+            for backoff_pid, retry_at in list(self._upload_backoff.items()):
+                if now_mono >= retry_at:
+                    del self._upload_backoff[backoff_pid]
+                    continue
+                backoff_printers.add(backoff_pid)
+                if backoff_pid in self._upload_refusal_notified:
+                    silent_hold_printers.add(backoff_pid)
+                else:
+                    self._upload_refusal_notified.add(backoff_pid)
+                mark_busy(
+                    backoff_pid,
+                    f"its file service refused the last upload; retrying in {retry_at - now_mono:.0f}s",
+                )
+                item_hold_reasons.setdefault(
+                    backoff_pid,
+                    f"{printer_label(backoff_pid)} is not accepting files — Bambuddy will retry automatically",
+                )
+
             # Printers held by a Home Assistant sensor interlock (#1148) — an
             # Printers held by a Home Assistant sensor interlock (#1148) — an
             # enclosure door left open, say. The fixed-printer branch turns
             # enclosure door left open, say. The fixed-printer branch turns
             # this into a waiting_reason the user can act on; the model-based
             # this into a waiting_reason the user can act on; the model-based
@@ -1762,6 +1834,7 @@ class PrintScheduler:
                         await hold_item(
                         await hold_item(
                             item,
                             item,
                             item_hold_reasons.get(item.printer_id) or f"Busy: {printer_label(item.printer_id)}",
                             item_hold_reasons.get(item.printer_id) or f"Busy: {printer_label(item.printer_id)}",
+                            notify=item.printer_id not in silent_hold_printers,
                         )
                         )
                         continue
                         continue
 
 
@@ -2196,7 +2269,9 @@ class PrintScheduler:
             # auxiliary check wedge the queue. The bed simply stays wherever it
             # auxiliary check wedge the queue. The bed simply stays wherever it
             # was, and the next tick tries again.
             # was, and the next tick tries again.
             try:
             try:
-                await self._apply_keep_warm(db, items, dispatch_ids, busy_printers, require_plate_clear)
+                await self._apply_keep_warm(
+                    db, items, dispatch_ids, busy_printers - backoff_printers, require_plate_clear
+                )
             except Exception as e:
             except Exception as e:
                 logger.warning("Keep-warm pass failed, continuing with dispatch: %s", e, exc_info=True)
                 logger.warning("Keep-warm pass failed, continuing with dispatch: %s", e, exc_info=True)
 
 
@@ -2387,6 +2462,59 @@ class PrintScheduler:
                 # dispatchable again on the next tick.
                 # dispatchable again on the next tick.
                 await self._clear_dispatch_claim(item_db, item_id)
                 await self._clear_dispatch_claim(item_db, item_id)
 
 
+    async def _requeue_after_upload_refused(
+        self,
+        db: AsyncSession,
+        item: PrintQueueItem,
+        printer: Printer,
+        error_msg: str,
+        toast_uid: int | None,
+    ) -> None:
+        """Put an item back in the queue after its file never reached the printer (#3210).
+
+        The item keeps its printer. Its AMS mapping was computed against that
+        printer's trays, and nothing on the row says whether a mapping was
+        computed or set by the user, so moving it to a sibling could print from
+        the wrong slots. The printer goes into `_upload_backoff` instead, which
+        keeps every other item away from it; this one waits there and is the
+        only thing that knocks again, once per backoff window. The window
+        grows with each refusal in a row; see ``_upload_backoff_seconds``.
+
+        `dispatch_attempts` is not charged: that budget bounds a printer that
+        takes the file and then never starts, which this is not.
+        """
+        refusals = self._upload_refusals.get(printer.id, 0) + 1
+        self._upload_refusals[printer.id] = refusals
+        backoff = _upload_backoff_seconds(refusals)
+        self._upload_backoff[printer.id] = time.monotonic() + backoff
+        # The row is still `pending` -- it only moves to `printing` after a
+        # successful upload -- so there is no status to write back. Writing one
+        # anyway would undo a cancel that landed during the upload.
+        if item.error_message:
+            item.error_message = None
+            await db.commit()
+        logger.warning(
+            "Queue item %s: upload to printer %s (%s) never reached it — %s Kept in the queue; "
+            "refusal %d in a row, so the printer is out of dispatch for %ds.",
+            item.id,
+            printer.id,
+            printer.name,
+            error_msg,
+            refusals,
+            backoff,
+        )
+        try:
+            # Closes the dispatch toast for this attempt. Without it the toast
+            # keeps spinning on an upload that has ended.
+            await ws_manager.send_queue_item_failed(
+                user_id=toast_uid,
+                queue_item_id=item.id,
+                printer_id=item.printer_id,
+                reason="upload_failed",
+            )
+        except Exception:
+            pass  # toast is best-effort
+
     def _rollback_unconfirmed_expected_print(self, item_id: int) -> None:
     def _rollback_unconfirmed_expected_print(self, item_id: int) -> None:
         """Drop an expectation for a print command that was never sent.
         """Drop an expectation for a print command that was never sent.
 
 
@@ -7697,6 +7825,10 @@ class PrintScheduler:
             # way to say so here, so the card got named even for a TLS
             # way to say so here, so the card got named even for a TLS
             # handshake that never reached the printer's filesystem (#2899).
             # handshake that never reached the printer's filesystem (#2899).
             error_msg = upload_error or describe_upload_failure(upload_failure.failure)
             error_msg = upload_error or describe_upload_failure(upload_failure.failure)
+            failure = upload_failure.failure
+            if upload_error is None and failure is not None and failure.kind in _UPLOAD_REQUEUE_KINDS:
+                await self._requeue_after_upload_refused(db, item, printer, error_msg, toast_uid)
+                return
             item.status = "failed"
             item.status = "failed"
             item.error_message = error_msg
             item.error_message = error_msg
             item.completed_at = datetime.now(timezone.utc)
             item.completed_at = datetime.now(timezone.utc)
@@ -7728,6 +7860,11 @@ class PrintScheduler:
             await self._power_off_if_needed(db, item)
             await self._power_off_if_needed(db, item)
             return
             return
 
 
+        # The printer took a file, so any run of refused uploads is over: the
+        # next refusal starts again at the shortest backoff, and notifies (#3210).
+        self._upload_refusals.pop(printer.id, None)
+        self._upload_refusal_notified.discard(printer.id)
+
         # Parse AMS mapping if stored
         # Parse AMS mapping if stored
         ams_mapping = None
         ams_mapping = None
         if item.ams_mapping:
         if item.ams_mapping:

+ 4 - 16
backend/tests/unit/services/test_upload_failure_reason_2899.py

@@ -400,22 +400,10 @@ async def _dispatch_failing_with(dispatch_case, failure: FtpFailure | None):
 
 
 
 
 class TestWhatTheQueueEntrySays:
 class TestWhatTheQueueEntrySays:
-    async def test_a_handshake_failure_does_not_send_anyone_to_the_sd_card(self, dispatch_case):
-        """The report's own case: a TLS failure, answered with card advice.
-
-        The reporter acted on it and restarted the printer. Nothing in that
-        path reaches the printer's filesystem, and the cool-off that made the
-        next dispatch fail identically lives in Bambuddy's memory, where
-        power-cycling a printer does not reach.
-        """
-        message, reason = await _dispatch_failing_with(
-            dispatch_case, FtpFailure(FtpFailureKind.HANDSHAKE, "WRONG_VERSION_NUMBER")
-        )
-
-        assert "inserted" not in message, message
-        assert "FAT32" not in message, message
-        assert "not with TLS" in message, message
-        assert reason == message
+    # A handshake failure used to be this class's lead case. Since #3210 it no
+    # longer fails the item at all -- the file never reached the printer, so the
+    # item goes back in the queue (test_scheduler_upload_requeue_3210.py). Its
+    # wording is still pinned by TestTheWording above.
 
 
     async def test_a_553_still_gets_the_card_advice(self, dispatch_case):
     async def test_a_553_still_gets_the_card_advice(self, dispatch_case):
         """The advice was written for this case and belongs to it.
         """The advice was written for this case and belongs to it.

+ 518 - 0
backend/tests/unit/test_scheduler_upload_requeue_3210.py

@@ -0,0 +1,518 @@
+"""A failed upload must not eat the queue (#3210).
+
+The reporter queued 20 jobs for "any P2S". One P2S's file service was out of
+connection slots -- it answered port 990 with ``421 There are too many
+connections from your internet address`` -- so every upload to it failed. Each
+failure marked the item ``failed``, which left that printer idle, so the next
+pass handed it the next item. One item died about every nine seconds: 43 jobs
+gone in ten minutes, none printed on that printer.
+
+Two changes, pinned here:
+
+* An upload whose file never reached the printer (handshake refused, cool-off,
+  timeout, dropped connection) puts the item back in the queue. Failures that
+  would repeat on every retry -- a rejected access code, a full card -- still
+  fail it.
+* The printer is out of dispatch for a backoff window, so nothing else is
+  sent to it, and the item that was put back waits there with a reason.
+"""
+
+import asyncio
+import time
+from contextlib import ExitStack, asynccontextmanager
+from pathlib import Path
+from types import SimpleNamespace
+from unittest.mock import AsyncMock, MagicMock, patch
+
+import pytest
+from sqlalchemy.ext.asyncio import async_sessionmaker, create_async_engine
+
+import backend.app.models  # noqa: F401 - populate Base.metadata
+import backend.app.services.archive as archive_module
+import backend.app.services.print_scheduler as scheduler_module
+from backend.app.core.database import Base
+from backend.app.models.archive import PrintArchive
+from backend.app.models.print_queue import PrintQueueItem
+from backend.app.models.printer import Printer
+from backend.app.services.bambu_ftp import BambuFTPClient, FtpFailure, FtpFailureKind, UploadCancelled
+from backend.app.services.print_scheduler import (
+    UPLOAD_FAILURE_BACKOFF_MAX_SECONDS,
+    UPLOAD_FAILURE_BACKOFF_SECONDS,
+    PrintScheduler,
+)
+
+pytestmark = pytest.mark.unit
+
+REFUSING_IP = "10.0.0.8"
+HEALTHY_IP = "10.0.0.9"
+
+
+@pytest.fixture
+async def farm(tmp_path):
+    """Two printers -- one whose file service refuses everything -- and an item factory."""
+    engine = create_async_engine("sqlite+aiosqlite:///:memory:", echo=False)
+    async with engine.begin() as conn:
+        await conn.run_sync(Base.metadata.create_all)
+    session_maker = async_sessionmaker(engine, expire_on_commit=False)
+
+    base_dir = tmp_path / "farm"
+    (base_dir / "archives").mkdir(parents=True, exist_ok=True)
+
+    async with session_maker() as db:
+        refusing = Printer(
+            name="P2S-8", serial_number="S8", ip_address=REFUSING_IP, access_code="12345678", model="P2S"
+        )
+        healthy = Printer(name="P2S-9", serial_number="S9", ip_address=HEALTHY_IP, access_code="12345678", model="P2S")
+        db.add_all([refusing, healthy])
+        await db.commit()
+        ids = SimpleNamespace(refusing=refusing.id, healthy=healthy.id)
+
+    counter = iter(range(1000))
+
+    async def add_item(*, printer_id: int | None = None, target_model: str | None = None) -> int:
+        n = next(counter)
+        async with session_maker() as db:
+            archive_rel = Path("archives") / f"job-{n}.3mf"
+            (base_dir / archive_rel).write_bytes(b"archive payload")
+            archive = PrintArchive(
+                printer_id=printer_id,
+                filename=f"job-{n}.3mf",
+                file_path=str(archive_rel),
+                file_size=15,
+                status="completed",
+            )
+            db.add(archive)
+            await db.flush()
+            item = PrintQueueItem(
+                printer_id=printer_id,
+                target_model=target_model,
+                archive_id=archive.id,
+                status="pending",
+                position=n,
+            )
+            db.add(item)
+            await db.commit()
+            return item.id
+
+    try:
+        yield SimpleNamespace(session_maker=session_maker, base_dir=base_dir, ids=ids, add_item=add_item)
+    finally:
+        await engine.dispose()
+
+
+def _upload_refused_by(ip: str, failure: FtpFailure):
+    """An upload that fails with *failure* on *ip* and succeeds everywhere else."""
+
+    async def _upload(ip_address, *_args, **kwargs):
+        if ip_address != ip:
+            return True
+        if kwargs.get("failure") is not None:
+            kwargs["failure"].failure = failure
+        return False
+
+    return _upload
+
+
+@asynccontextmanager
+async def _scheduler(
+    ctx,
+    upload,
+    *,
+    busy: set[int] | None = None,
+    scheduler: PrintScheduler | None = None,
+    waiting_notify: AsyncMock | None = None,
+):
+    """A scheduler wired to *ctx*'s database, with every printer idle unless in *busy*.
+
+    Pass the same *waiting_notify* to several passes to count notifications
+    across them.
+    """
+    scheduler = scheduler or PrintScheduler()
+    busy = busy if busy is not None else set()
+    failed_notify = AsyncMock()
+    waiting_notify = waiting_notify or AsyncMock()
+
+    def _real_spawn(coro, *, name=None):
+        return asyncio.create_task(coro, name=name)
+
+    patches = [
+        patch.object(scheduler_module.settings, "base_dir", ctx.base_dir),
+        patch.object(archive_module.settings, "base_dir", ctx.base_dir),
+        patch.object(archive_module.settings, "archive_dir", ctx.base_dir / "archive"),
+        patch("backend.app.services.print_scheduler.async_session", ctx.session_maker),
+        patch("backend.app.core.database.async_session", ctx.session_maker),
+        patch("backend.app.services.print_scheduler.printer_manager.is_connected", MagicMock(return_value=True)),
+        patch(
+            "backend.app.services.print_scheduler.printer_manager.get_status",
+            MagicMock(return_value=SimpleNamespace(state="IDLE", subtask_id=None, gcode_file=None, raw_data={})),
+        ),
+        patch(
+            "backend.app.services.print_scheduler.printer_manager.is_awaiting_plate_clear",
+            MagicMock(return_value=False),
+        ),
+        patch("backend.app.services.print_scheduler.printer_manager.start_print", MagicMock(return_value=True)),
+        patch("backend.app.services.print_scheduler.printer_manager.set_awaiting_plate_clear", MagicMock()),
+        patch("backend.app.services.print_scheduler.upload_file_async", upload),
+        patch("backend.app.services.print_scheduler.delete_file_async", AsyncMock(return_value=True)),
+        patch(
+            "backend.app.services.print_scheduler.get_ftp_retry_settings",
+            AsyncMock(return_value=(False, 0, 0, 1.0)),
+        ),
+        patch("backend.app.services.print_scheduler.cache_3mf_download", MagicMock()),
+        patch("backend.app.services.print_scheduler.spawn_background_task", _real_spawn),
+        patch("backend.app.services.notification_service.notification_service.on_queue_job_started", AsyncMock()),
+        patch("backend.app.services.notification_service.notification_service.on_queue_job_failed", failed_notify),
+        patch("backend.app.services.notification_service.notification_service.on_queue_job_assigned", AsyncMock()),
+        patch("backend.app.services.notification_service.notification_service.on_queue_job_waiting", waiting_notify),
+        patch("backend.app.services.mqtt_relay.mqtt_relay.on_queue_job_started", AsyncMock()),
+        patch.object(scheduler, "_is_printer_idle", MagicMock(side_effect=lambda pid, *_a, **_k: pid not in busy)),
+        patch.object(scheduler, "_ensure_ams_mapping", AsyncMock(return_value=None)),
+        patch.object(scheduler, "_block_on_filament_deficit", AsyncMock(return_value=False)),
+        patch.object(scheduler, "_propagate_owner_to_printer_manager", AsyncMock()),
+        patch.object(scheduler, "_power_off_if_needed", AsyncMock()),
+        patch.object(scheduler, "_preheat_and_soak", AsyncMock()),
+        patch.object(scheduler, "_check_auto_drying", AsyncMock()),
+        patch.object(scheduler, "_watchdog_print_start", AsyncMock()),
+    ]
+    with ExitStack() as stack:
+        for patcher in patches:
+            stack.enter_context(patcher)
+        scheduler.failed_notify = failed_notify
+        yield scheduler
+        tasks = [task for (task, _pid) in scheduler._inflight.values()]
+        if tasks:
+            await asyncio.gather(*tasks, return_exceptions=True)
+
+
+async def _item(ctx, item_id: int) -> PrintQueueItem:
+    async with ctx.session_maker() as db:
+        return await db.get(PrintQueueItem, item_id)
+
+
+HANDSHAKE = FtpFailure(FtpFailureKind.HANDSHAKE, "WRONG_VERSION_NUMBER (printer answered in cleartext: 421 ...)")
+
+
+# ---------------------------------------------------------------------------
+# What one failed upload does to its item
+# ---------------------------------------------------------------------------
+class TestOneFailedUpload:
+    @pytest.mark.asyncio
+    @pytest.mark.parametrize(
+        "kind",
+        [FtpFailureKind.HANDSHAKE, FtpFailureKind.COOLOFF, FtpFailureKind.TIMEOUT, FtpFailureKind.NETWORK],
+    )
+    async def test_a_file_that_never_arrived_keeps_the_item(self, farm, kind):
+        item_id = await farm.add_item(printer_id=farm.ids.refusing)
+        upload = _upload_refused_by(REFUSING_IP, FtpFailure(kind, "detail"))
+
+        async with _scheduler(farm, upload) as scheduler:
+            await scheduler.check_queue()
+
+        item = await _item(farm, item_id)
+        assert item.status == "pending"
+        assert item.printer_id == farm.ids.refusing
+        assert item.error_message is None
+        assert item.dispatch_attempts == 0, "the start-watchdog budget is for a printer that took the file"
+        scheduler.failed_notify.assert_not_awaited()
+        assert farm.ids.refusing in scheduler._upload_backoff
+
+    @pytest.mark.asyncio
+    @pytest.mark.parametrize(
+        "failure",
+        [
+            FtpFailure(FtpFailureKind.AUTH, "530 Login incorrect.", "530"),
+            FtpFailure(FtpFailureKind.STORAGE, "553 Could not create file.", "553"),
+            FtpFailure(FtpFailureKind.NOT_FOUND, "550 Permission denied.", "550"),
+            FtpFailure(FtpFailureKind.UNKNOWN, "500 what"),
+            None,
+        ],
+        ids=["auth", "storage", "not_found", "unknown", "unreported"],
+    )
+    async def test_a_failure_that_would_repeat_still_fails_the_item(self, farm, failure):
+        """Retrying a wrong access code or a full card every five minutes helps no one."""
+        item_id = await farm.add_item(printer_id=farm.ids.refusing)
+
+        async def upload(*_args, **kwargs):
+            if failure is not None:
+                kwargs["failure"].failure = failure
+            return False
+
+        async with _scheduler(farm, upload) as scheduler:
+            await scheduler.check_queue()
+
+        item = await _item(farm, item_id)
+        assert item.status == "failed"
+        assert item.error_message
+        scheduler.failed_notify.assert_awaited_once()
+        assert farm.ids.refusing not in scheduler._upload_backoff
+
+    @pytest.mark.asyncio
+    async def test_an_upload_that_overran_its_deadline_still_fails(self, farm):
+        """A link too slow to finish would be just as slow next time (#2529)."""
+        item_id = await farm.add_item(printer_id=farm.ids.refusing)
+
+        async def upload(*_args, **kwargs):
+            kwargs["failure"].failure = FtpFailure(FtpFailureKind.TIMEOUT, "deadline")
+            raise UploadCancelled("too slow")
+
+        async with _scheduler(farm, upload) as scheduler:
+            await scheduler.check_queue()
+
+        assert (await _item(farm, item_id)).status == "failed"
+
+    @pytest.mark.asyncio
+    async def test_a_cancel_during_the_upload_is_not_undone(self, farm):
+        """The row is never written back to pending, so a cancel that won stays won."""
+        item_id = await farm.add_item(printer_id=farm.ids.refusing)
+
+        async def upload(*_args, **kwargs):
+            async with farm.session_maker() as other:
+                row = await other.get(PrintQueueItem, item_id)
+                row.status = "cancelled"
+                await other.commit()
+            kwargs["failure"].failure = HANDSHAKE
+            return False
+
+        async with _scheduler(farm, upload) as scheduler:
+            await scheduler.check_queue()
+
+        assert (await _item(farm, item_id)).status == "cancelled"
+
+
+# ---------------------------------------------------------------------------
+# The report: a refusing printer must not drain the queue
+# ---------------------------------------------------------------------------
+class TestTheQueueIsNotDrained:
+    @pytest.mark.asyncio
+    async def test_any_model_items_stop_going_to_the_refusing_printer(self, farm):
+        """#3210's shape: "any P2S" items, one P2S refusing every upload.
+
+        Pass 1 sends one item to each printer; the refusing one keeps its item.
+        Pass 2 has the healthy printer busy printing and the refusing one in
+        backoff, so the rest wait instead of being fed to it one by one.
+        """
+        ids = [await farm.add_item(target_model="P2S") for _ in range(5)]
+        upload = _upload_refused_by(REFUSING_IP, HANDSHAKE)
+        scheduler = PrintScheduler()
+
+        async with _scheduler(farm, upload, scheduler=scheduler):
+            await scheduler.check_queue()
+        async with _scheduler(farm, upload, scheduler=scheduler, busy={farm.ids.healthy}):
+            for _ in range(3):
+                await scheduler.check_queue()
+
+        items = [await _item(farm, i) for i in ids]
+        assert [i.status for i in items].count("failed") == 0, [(i.id, i.status, i.error_message) for i in items]
+        assert [i.status for i in items].count("printing") == 1
+        held = [i for i in items if i.printer_id == farm.ids.refusing]
+        assert len(held) == 1, "exactly one item waits on the refusing printer"
+        assert held[0].status == "pending"
+        waiting = [i for i in items if i.status == "pending" and i.printer_id is None]
+        assert len(waiting) == 3, "the rest stay unassigned, free for any printer that comes free"
+
+    @pytest.mark.asyncio
+    async def test_the_item_kept_on_the_printer_says_why(self, farm):
+        item_id = await farm.add_item(printer_id=farm.ids.refusing)
+        upload = _upload_refused_by(REFUSING_IP, HANDSHAKE)
+        scheduler = PrintScheduler()
+
+        async with _scheduler(farm, upload, scheduler=scheduler):
+            await scheduler.check_queue()
+        async with _scheduler(farm, upload, scheduler=scheduler):
+            await scheduler.check_queue()
+
+        item = await _item(farm, item_id)
+        assert item.status == "pending"
+        assert item.waiting_reason == "P2S-8 is not accepting files — Bambuddy will retry automatically"
+
+    @pytest.mark.asyncio
+    async def test_the_printer_is_tried_again_once_the_backoff_ends(self, farm):
+        item_id = await farm.add_item(printer_id=farm.ids.refusing)
+        scheduler = PrintScheduler()
+
+        async with _scheduler(farm, _upload_refused_by(REFUSING_IP, HANDSHAKE), scheduler=scheduler):
+            await scheduler.check_queue()
+
+        retry_at = scheduler._upload_backoff[farm.ids.refusing]
+        assert retry_at - time.monotonic() == pytest.approx(UPLOAD_FAILURE_BACKOFF_SECONDS, abs=5)
+
+        # The printer recovered and the window has passed.
+        scheduler._upload_backoff[farm.ids.refusing] = time.monotonic() - 1
+        async with _scheduler(farm, AsyncMock(return_value=True), scheduler=scheduler):
+            await scheduler.check_queue()
+
+        assert (await _item(farm, item_id)).status == "printing"
+        assert farm.ids.refusing not in scheduler._upload_backoff
+
+
+# ---------------------------------------------------------------------------
+# A printer that stays broken: retries thin out, and say so once
+# ---------------------------------------------------------------------------
+def _expire_backoff(scheduler: PrintScheduler, printer_id: int) -> None:
+    """Jump past the printer's backoff window, as if the time had passed."""
+    scheduler._upload_backoff[printer_id] = time.monotonic() - 1
+
+
+def _remaining(scheduler: PrintScheduler, printer_id: int) -> float:
+    return scheduler._upload_backoff[printer_id] - time.monotonic()
+
+
+async def _finish(ctx, scheduler: PrintScheduler, item_id: int) -> None:
+    """Mark a dispatched item's print done, so its printer can take the next one."""
+    async with ctx.session_maker() as db:
+        row = await db.get(PrintQueueItem, item_id)
+        assert row.status == "printing"
+        row.status = "completed"
+        await db.commit()
+        scheduler._release_dispatch_hold(row.printer_id)
+
+
+class TestAPrinterThatStaysBroken:
+    @pytest.mark.asyncio
+    async def test_each_refusal_in_a_row_doubles_the_wait_up_to_the_cap(self, farm):
+        """#3210's printer refused for about 40 hours.
+
+        At a flat five minutes that is ~480 retries, each one a preheat cycle
+        where preheat is on, five connection attempts and a page of log.
+        """
+        await farm.add_item(printer_id=farm.ids.refusing)
+        upload = _upload_refused_by(REFUSING_IP, HANDSHAKE)
+        scheduler = PrintScheduler()
+
+        waits = []
+        for _ in range(6):
+            async with _scheduler(farm, upload, scheduler=scheduler):
+                await scheduler.check_queue()
+            waits.append(_remaining(scheduler, farm.ids.refusing))
+            _expire_backoff(scheduler, farm.ids.refusing)
+
+        expected = [300, 600, 1200, 2400, 3600, 3600]
+        assert expected[0] == UPLOAD_FAILURE_BACKOFF_SECONDS
+        assert expected[-1] == UPLOAD_FAILURE_BACKOFF_MAX_SECONDS
+        assert waits == [pytest.approx(w, abs=5) for w in expected]
+
+    @pytest.mark.asyncio
+    async def test_a_successful_upload_resets_the_wait(self, farm):
+        first = await farm.add_item(printer_id=farm.ids.refusing)
+        refused = _upload_refused_by(REFUSING_IP, HANDSHAKE)
+        scheduler = PrintScheduler()
+
+        for _ in range(3):
+            async with _scheduler(farm, refused, scheduler=scheduler):
+                await scheduler.check_queue()
+            _expire_backoff(scheduler, farm.ids.refusing)
+        async with _scheduler(farm, AsyncMock(return_value=True), scheduler=scheduler):
+            await scheduler.check_queue()
+        assert farm.ids.refusing not in scheduler._upload_refusals
+        await _finish(farm, scheduler, first)
+
+        # It breaks again later: back to the first window, not the fourth.
+        await farm.add_item(printer_id=farm.ids.refusing)
+        async with _scheduler(farm, refused, scheduler=scheduler, busy=set()):
+            await scheduler.check_queue()
+        assert _remaining(scheduler, farm.ids.refusing) == pytest.approx(UPLOAD_FAILURE_BACKOFF_SECONDS, abs=5)
+
+    @pytest.mark.asyncio
+    async def test_one_outage_sends_one_waiting_notification(self, farm):
+        """Every retry clears the waiting reason and the next refusal sets it again.
+
+        `hold_item` reads that as a new reason each time, and "Job Waiting" is
+        on by default -- so without a guard a 40-hour outage notifies ~480 times.
+        """
+        await farm.add_item(printer_id=farm.ids.refusing)
+        upload = _upload_refused_by(REFUSING_IP, HANDSHAKE)
+        scheduler = PrintScheduler()
+        waiting = AsyncMock()
+
+        for _ in range(4):
+            # The pass that dispatches and is refused...
+            async with _scheduler(farm, upload, scheduler=scheduler, waiting_notify=waiting):
+                await scheduler.check_queue()
+            # ...and the pass that holds the item while the printer waits.
+            async with _scheduler(farm, upload, scheduler=scheduler, waiting_notify=waiting):
+                await scheduler.check_queue()
+            _expire_backoff(scheduler, farm.ids.refusing)
+
+        assert waiting.await_count == 1
+        assert "not accepting files" in waiting.await_args.kwargs["waiting_reason"]
+
+    @pytest.mark.asyncio
+    async def test_a_new_outage_after_a_recovery_notifies_again(self, farm):
+        first = await farm.add_item(printer_id=farm.ids.refusing)
+        refused = _upload_refused_by(REFUSING_IP, HANDSHAKE)
+        scheduler = PrintScheduler()
+        waiting = AsyncMock()
+
+        async def outage():
+            async with _scheduler(farm, refused, scheduler=scheduler, waiting_notify=waiting):
+                await scheduler.check_queue()
+            async with _scheduler(farm, refused, scheduler=scheduler, waiting_notify=waiting):
+                await scheduler.check_queue()
+            _expire_backoff(scheduler, farm.ids.refusing)
+
+        await outage()
+        async with _scheduler(farm, AsyncMock(return_value=True), scheduler=scheduler, waiting_notify=waiting):
+            await scheduler.check_queue()
+        await _finish(farm, scheduler, first)
+        await farm.add_item(printer_id=farm.ids.refusing)
+        await outage()
+
+        assert waiting.await_count == 2
+
+    @pytest.mark.asyncio
+    async def test_keep_warm_does_not_hold_a_bed_for_a_printer_in_backoff(self, farm):
+        """There is no print coming to keep the bed warm for.
+
+        The printer is in busy_printers and has a pending item, which is
+        exactly what keep-warm looks for -- and each retry would restart its
+        time cap, so the bed could stay hot for the whole outage.
+        """
+        await farm.add_item(printer_id=farm.ids.refusing)
+        upload = _upload_refused_by(REFUSING_IP, HANDSHAKE)
+        scheduler = PrintScheduler()
+
+        async with _scheduler(farm, upload, scheduler=scheduler):
+            await scheduler.check_queue()
+        keep_warm = AsyncMock()
+        async with _scheduler(farm, upload, scheduler=scheduler):
+            with patch.object(scheduler, "_apply_keep_warm", keep_warm):
+                await scheduler.check_queue()
+
+        busy_seen = keep_warm.await_args.args[3]
+        assert farm.ids.refusing not in busy_seen
+
+
+# ---------------------------------------------------------------------------
+# 452 is the card, not the network
+# ---------------------------------------------------------------------------
+class TestA452IsAStorageReply:
+    def _upload_raising(self, error, tmp_path):
+        local = tmp_path / "job.3mf"
+        local.write_bytes(b"x" * 16)
+        client = BambuFTPClient(REFUSING_IP, "12345678")
+        client._ftp = MagicMock()
+        client._ftp.transfercmd.side_effect = error
+        assert client.upload_file(local, "/job.3mf") is False
+        return client.last_failure
+
+    def test_452_is_storage(self, tmp_path):
+        """ftplib raises it as error_temp, but it says the card is full.
+
+        Read as NETWORK it would be retried forever instead of failing with
+        the advice to check the card.
+        """
+        import ftplib  # nosec B402 -- tests construct real ftplib error types
+
+        failure = self._upload_raising(ftplib.error_temp("452 Insufficient storage space."), tmp_path)
+        assert failure.kind is FtpFailureKind.STORAGE
+        assert failure.code == "452"
+
+    def test_other_transient_replies_are_still_network(self, tmp_path):
+        import ftplib  # nosec B402 -- tests construct real ftplib error types
+
+        failure = self._upload_raising(ftplib.error_temp("421 There are too many connections."), tmp_path)
+        assert failure.kind is FtpFailureKind.NETWORK
+
+    def test_a_socket_error_is_never_read_as_a_reply_code(self, tmp_path):
+        failure = self._upload_raising(OSError("452 looks like a code but is not a reply"), tmp_path)
+        assert failure.kind is FtpFailureKind.NETWORK