[Devel] [PATCH VZ10 v4 07/11] drivers/md/dm-qcow2: update metadata on subclusters discard

Andrey Zhadchenko andrey.zhadchenko at virtuozzo.com
Thu Aug 27 19:06:15 MSK 2026


On images with extended L2 entries a discard smaller than the whole
cluster only punches a hole in the image file, while the subcluster
allocation bitmap still reports the range as allocated.

Extend prepare_cluster_discard() to also handle discards covering
whole subclusters of a mapped non-compressed cluster: clear the
"allocated" bits of the covered subclusters in the extended L2 entry
via the existing only_set_ext_l2 machinery. If the range may be
mapped in lower delta, set the "reads as zeroes" bits instead, so the
discarded range doesn't expose stale lower data. The cluster itself
remains allocated, so refcounts are not touched and the unuse stage
has nothing to do (empty unuse range). The covered range is hole
punched unless the cluster is shared with a snapshot.

ext_l2 and unuse range handling are in different if blocks for a while:
next patch will cover the case when we need uncoupling.

Feature: dm-qcow2: block device over QCOW2 files driver
https://virtuozzo.atlassian.net/browse/VSTOR-139406
Signed-off-by: Andrey Zhadchenko <andrey.zhadchenko at virtuozzo.com>
---
 drivers/md/dm-qcow2-map.c | 69 ++++++++++++++++++++++++++++++++-------
 1 file changed, 58 insertions(+), 11 deletions(-)

diff --git a/drivers/md/dm-qcow2-map.c b/drivers/md/dm-qcow2-map.c
index 52a630685518b..609cdd8837b8d 100644
--- a/drivers/md/dm-qcow2-map.c
+++ b/drivers/md/dm-qcow2-map.c
@@ -107,6 +107,21 @@ static bool qio_covers_full_clu(struct qcow2 *qcow2, struct qio *qio)
 	       qio->bi_iter.bi_size == qcow2->clu_size;
 }
 
+/* Mask of subclusters fully covered by qio (qio is trimmed inward) */
+static u32 qio_full_subclus_mask(struct qcow2 *qcow2, struct qio *qio)
+{
+	u32 off = bytes_off_in_cluster(qcow2, qio);
+	u32 first_bit = DIV_ROUND_UP(off, qcow2->subclu_size);
+	u32 end_bit = (off + qio->bi_iter.bi_size) / qcow2->subclu_size;
+
+	WARN_ON_ONCE(!qcow2->ext_l2);
+
+	if (end_bit <= first_bit)
+		return 0;
+
+	return GENMASK(end_bit - 1, first_bit);
+}
+
 static loff_t compressed_clu_end_pos(loff_t start, sector_t compressed_sectors)
 {
 	if (start % SECTOR_SIZE == 0)
@@ -3357,11 +3372,21 @@ static void issue_discard(struct qcow2_map *map, struct qio *qio)
 static bool qio_discard_updates_metadata(struct qcow2 *qcow2, struct qio *qio,
 					 struct qcow2_map *map)
 {
+	u32 mask;
+
 	if (!(map->level & L2_LEVEL) || !map->data_clu_alloced)
 		return false;
 	if (qio_covers_full_clu(qcow2, qio))
 		return true;
-	return false;
+	if (!qcow2->ext_l2 || map->compressed)
+		return false;
+
+	mask = qio_full_subclus_mask(qcow2, qio);
+	if ((u32)map->ext_l2 & mask)
+		return true;
+
+	return maybe_mapped_in_lower_delta(qcow2, qio) &&
+	       (mask & ~(u32)(map->ext_l2 >> 32));
 }
 
 /*
@@ -3371,6 +3396,9 @@ static bool qio_discard_updates_metadata(struct qcow2 *qcow2, struct qio *qio,
  *
  * Discard covering the whole cluster replaces the entry and unuses
  * the discarded (or COW source) cluster.
+ * Subclusters fully covered by the discard are marked unallocated in
+ * ext_l2 bitmap: the cluster itself remains allocated, so refcounts
+ * are not touched (hence empty unuse range).
  * If the backing is present, set 'reads as zeroes' to avoid exposing
  * stale data.
  */
@@ -3381,6 +3409,7 @@ static int prepare_cluster_discard(struct qcow2 *qcow2, struct qio **qio,
 	u32 index_in_page = map->l2.index_in_page;
 	struct md_page *md = map->l2.md;
 	loff_t unuse_pos, unuse_end;
+	bool whole_clu;
 	struct qio_ext *ext;
 	u64 new_ext_l2 = 0;
 	int ret;
@@ -3397,17 +3426,33 @@ static int prepare_cluster_discard(struct qcow2 *qcow2, struct qio **qio,
 	}
 	spin_unlock_irq(&qcow2->md_pages_lock);
 
-	if (map->clu_is_cow) {
-		/* Cluster is shared or compressed. Decrement refcount. */
-		unuse_pos = map->cow_clu_pos;
-		unuse_end = map->cow_clu_end;
+	whole_clu = qio_covers_full_clu(qcow2, *qio);
+	if (whole_clu) {
+		if (zeroes && qcow2->ext_l2)
+			new_ext_l2 = (u64)U32_MAX << 32;
+	} else {
+		u64 mask = qio_full_subclus_mask(qcow2, *qio);
+
+		new_ext_l2 = map->ext_l2 & ~(mask << 32 | mask);
+		if (zeroes)
+			new_ext_l2 |= mask << 32;
+	}
+
+	if (whole_clu) {
+		if (map->clu_is_cow) {
+			/* Cluster is shared or compressed. Decrement refcount. */
+			unuse_pos = map->cow_clu_pos;
+			unuse_end = map->cow_clu_end;
+		} else {
+			/* Nobody else refers the cluster: unuse it */
+			unuse_pos = map->data_clu_pos;
+			unuse_end = map->data_clu_pos + qcow2->clu_size;
+		}
 	} else {
-		/* Nobody else refers the cluster: unuse it after discard */
-		unuse_pos = map->data_clu_pos;
-		unuse_end = map->data_clu_pos + qcow2->clu_size;
+		/* Subclusters become unallocated, cluster remains */
+		unuse_pos = 0;
+		unuse_end = 0;
 	}
-	if (zeroes && qcow2->ext_l2)
-		new_ext_l2 = (u64)U32_MAX << 32;
 
 	ret = prepare_l_entry_replace(qcow2, map, *qio, md, index_in_page,
 				      unuse_pos, unuse_end, L2_LEVEL);
@@ -3418,7 +3463,9 @@ static int prepare_cluster_discard(struct qcow2 *qcow2, struct qio **qio,
 	ext->lx_md = md;
 
 	ext->new_ext_l2 = new_ext_l2;
-	if (zeroes && !qcow2->ext_l2)
+	if (!whole_clu)
+		ext->only_set_ext_l2 = true;
+	else if (zeroes && !qcow2->ext_l2)
 		ext->set_all_zeroes = true;
 
 	(*qio)->flags |= QIO_IS_DISCARD_FL;
-- 
2.43.5



More information about the Devel mailing list