public inbox for pve-devel@lists.proxmox.com
 help / color / mirror / Atom feed
From: Lukas Sichert <l.sichert@proxmox.com>
To: pve-devel@lists.proxmox.com
Cc: Lukas Sichert <l.sichert@proxmox.com>
Subject: [PATCH storage v10 1/6] lvm: saferemove: keep LVs where zero-out failed for manual zero-out
Date: Tue, 21 Jul 2026 14:37:16 +0200	[thread overview]
Message-ID: <20260721123724.45395-2-l.sichert@proxmox.com> (raw)
In-Reply-To: <20260721123724.45395-1-l.sichert@proxmox.com>

Currently even if 'zeroout' fails, the LV is removed and can't be zeroed
out manually later.

Let zeroing errors propagate from the secure delete command, and rename
failed removals to a 'failed-<N>-del-*' LV name instead of immediately
removing them.

Signed-off-by: Lukas Sichert <l.sichert@proxmox.com>
---
 src/PVE/Storage/LVMPlugin.pm | 119 ++++++++++++++++++++++++++++-------
 1 file changed, 95 insertions(+), 24 deletions(-)

diff --git a/src/PVE/Storage/LVMPlugin.pm b/src/PVE/Storage/LVMPlugin.pm
index a313ecc..4734e11 100644
--- a/src/PVE/Storage/LVMPlugin.pm
+++ b/src/PVE/Storage/LVMPlugin.pm
@@ -7,8 +7,10 @@ use Cwd qw(abs_path);
 use File::Basename;
 use IO::File;
 use JSON;
+use List::Util qw(max);
 
 use PVE::JSONSchema qw(get_standard_option);
+use PVE::RESTEnvironment qw(log_warn);
 use PVE::Tools qw(run_command file_read_firstline trim);
 
 use PVE::Storage::Common;
@@ -279,6 +281,49 @@ sub lvm_list_volumes {
     return $lvs;
 }
 
+my sub rename_after_failed_cleanup {
+    my ($class, $scfg, $storeid, $vg, $name) = @_;
+
+    eval {
+        my $failed_name;
+        $class->cluster_lock_storage(
+            $storeid,
+            $scfg->{shared},
+            undef,
+            sub {
+                my $vgs = lvm_vgs();
+                die "volume group '$vg' not found\n"
+                    if !defined($vgs->{$vg});
+
+                my $lvs = lvm_list_volumes($vg);
+                my $existing = $lvs->{$vg} // {};
+
+                my $prefix = 'failed-';
+                my $suffix = "-del-$name";
+
+                my $last_fail = max(
+                    -1,
+                    map {
+                        /^\Q$prefix\E(\d+)\Q$suffix\E$/ ? $1 : ()
+                    } keys %$existing,
+                );
+
+                $failed_name = 'failed-' . ($last_fail + 1) . $suffix;
+
+                my $cmd = ['/sbin/lvrename', $vg, "del-$name", $failed_name];
+                run_command(
+                    $cmd,
+                    errmsg => "lvrename '$vg/del-$name' to '$vg/$failed_name' error",
+                );
+                print "renamed '$vg/del-$name' to '$vg/$failed_name'\n";
+            },
+        );
+    };
+    if (my $rename_err = $@) {
+        print STDERR "ERROR: unable to rename '$vg/del-$name': $rename_err";
+    }
+}
+
 my sub free_lvm_volumes_locked {
     my ($class, $scfg, $storeid, $volnames) = @_;
 
@@ -327,6 +372,9 @@ my sub free_lvm_volumes_locked {
                 '-t',
                 "$throughput",
             ];
+            # FIXME: handle cstream's expected ENOSPC failure explicitly and let other
+            # errors propagate. For now, preserve the old behavior where cstream can
+            # fail successfully with ENOSPC after writing until the device is full.
             eval {
                 run_command(
                     $cmd,
@@ -345,41 +393,64 @@ my sub free_lvm_volumes_locked {
             }
 
             my $cmd = ['blkdiscard', $lvmpath, '-v', '--zeroout', '--step', "${stepsize}"];
-            eval { run_command($cmd); };
-            warn $@ if $@;
+            run_command($cmd);
         }
     };
 
     # we need to zero out LVM data for security reasons
     # and to allow thin provisioning
     my $zero_out_worker = sub {
+
+        my $total_cleanup_errors = 0;
         for my $name (@$volnames) {
             my $lvmpath = "/dev/$vg/del-$name";
             print "zero-out data on image $name ($lvmpath)\n";
 
-            my $cmd_activate = ['/sbin/lvchange', '-aly', $lvmpath];
-            run_command(
-                $cmd_activate,
-                errmsg => "can't activate LV '$lvmpath' to zero-out its data",
-            );
-            $cmd_activate = ['/sbin/lvchange', '--refresh', $lvmpath];
-            run_command(
-                $cmd_activate,
-                errmsg => "can't refresh LV '$lvmpath' to zero-out its data",
-            );
-
-            $secure_delete_cmd->($lvmpath);
+            eval {
+                # pass an errfunc here so that debug information is not by lvm to stderr,
+                # but by the print STDERR below with additional information
+                my $cmd_activate = ['/sbin/lvchange', '-aly', $lvmpath];
+                run_command(
+                    $cmd_activate,
+                    errmsg => "can't activate LV '$lvmpath' to zero-out its data",
+                    errfunc => sub { },
+                );
+                $cmd_activate = ['/sbin/lvchange', '--refresh', $lvmpath];
+                run_command(
+                    $cmd_activate,
+                    errmsg => "can't refresh LV '$lvmpath' to zero-out its data",
+                    errfunc => sub { },
+                );
+            };
+            if (my $activation_err = $@) {
+                print STDERR "ERROR: $activation_err";
+                eval { rename_after_failed_cleanup($class, $scfg, $storeid, $vg, $name) };
+                $total_cleanup_errors += 1;
+                next;
+            }
 
-            $class->cluster_lock_storage(
-                $storeid,
-                $scfg->{shared},
-                undef,
-                sub {
-                    my $cmd = ['/sbin/lvremove', '-f', "$vg/del-$name"];
-                    run_command($cmd, errmsg => "lvremove '$vg/del-$name' error");
-                },
-            );
-            print "successfully removed volume $name ($vg/del-$name)\n";
+            eval { $secure_delete_cmd->($lvmpath); };
+            if (my $cleanup_err = $@) {
+                print STDERR "ERROR: cleanup failed for lv $name: $cleanup_err";
+                eval { rename_after_failed_cleanup($class, $scfg, $storeid, $vg, $name) };
+                $total_cleanup_errors += 1;
+                next;
+            } else {
+                $class->cluster_lock_storage(
+                    $storeid,
+                    $scfg->{shared},
+                    undef,
+                    sub {
+                        my $cmd = ['/sbin/lvremove', '-f', "$vg/del-$name"];
+                        run_command($cmd, errmsg => "lvremove '$vg/del-$name' error");
+                    },
+                );
+                print "successfully removed volume $name ($vg/del-$name)\n";
+            }
+        }
+        if ($total_cleanup_errors != 0) {
+            my $number_of_vols = scalar @$volnames;
+            die "cleanup failed for $total_cleanup_errors out of $number_of_vols volumes\n";
         }
     };
 
-- 
2.47.3





  reply	other threads:[~2026-07-21 12:38 UTC|newest]

Thread overview: 7+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-07-21 12:37 [PATCH docs/manager/storage v10 0/6] fix #7339: lvmthick: add option to free storage for deleted VMs Lukas Sichert
2026-07-21 12:37 ` Lukas Sichert [this message]
2026-07-21 12:37 ` [PATCH storage v10 2/6] lvm: saferemove: zero out volumes range by range Lukas Sichert
2026-07-21 12:37 ` [PATCH storage v10 3/6] lvm: saferemove: make throughput an integer property Lukas Sichert
2026-07-21 12:37 ` [PATCH storage v10 4/6] fix #7339: lvm: add discard action for removed volumes Lukas Sichert
2026-07-21 12:37 ` [PATCH manager v10 5/6] fix #7339: lvm: add discard-on-remove option to UI Lukas Sichert
2026-07-21 12:37 ` [PATCH docs v10 6/6] fix #7339: lvm: document discard option Lukas Sichert

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20260721123724.45395-2-l.sichert@proxmox.com \
    --to=l.sichert@proxmox.com \
    --cc=pve-devel@lists.proxmox.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox
Service provided by Proxmox Server Solutions GmbH | Privacy | Legal