[Devel] [PATCH VZ10 6/6] drivers/md/dm-qcow2: seek unallocated L2 entries in one pass within md

Andrey Zhadchenko andrey.zhadchenko at virtuozzo.com
Wed Aug 12 15:50:20 MSK 2026



On 8/12/26 14:42, Vasileios Almpanis wrote:
> 
> On 8/10/26 2:30 PM, Andrey Zhadchenko wrote:
>> When a cluster is unallocated but its L2 table exists, the seek code
>> walks it one cluster at a time: SEEK_DATA re-parses metadata for
>> every cluster of the range, and with a backing image each cluster
>> additionally costs a qio and completion allocation for the lower
>> delta descent.
>>
>> If the unmapped range reaches the cluster border, prolong it over the
>> following unallocated L2 entries: they are cached in the very md page
>> the current entry was parsed from, so this is just scanning for
>> non-zero words. Then the lower delta is scanned in one pass over the
>> whole range, and SEEK_DATA without a backing image skips it in one go.
>>
>> Entries with any subcluster or "reads as zeroes" bits set stop the
>> scan and spin parse_metadata() machinery.
>>
>> On a 16G image (128K clusters, extended L2) with all L2 tables
>> allocated but no clusters mapped over a backing image with data at
>> 15G, warm-cache SEEK_DATA improves from 44.8 ms to 0.4 ms.
>>
>> https://virtuozzo.atlassian.net/browse/VSTOR-139407
>> Feature: dm-qcow2: block device over QCOW2 files driver
>> Signed-off-by: Andrey Zhadchenko <andrey.zhadchenko at virtuozzo.com>
>> ---
>>   drivers/md/dm-qcow2-map.c | 44 +++++++++++++++++++++++++++++++++++++--
>>   1 file changed, 42 insertions(+), 2 deletions(-)
>>
>> diff --git a/drivers/md/dm-qcow2-map.c b/drivers/md/dm-qcow2-map.c
>> index 6a73c32689eca..3bda108db725f 100644
>> --- a/drivers/md/dm-qcow2-map.c
>> +++ b/drivers/md/dm-qcow2-map.c
>> @@ -4441,6 +4441,34 @@ static inline void seek_qio_next_clu(struct qio 
>> *qio, struct qcow2_map *map)
>>       seek_qio_set_sector(qio, bi_sector);
>>   }
>> +/*
>> + * Manually scan L2 metadata pages until we hit something interesting.
>> + * We do not take md_pages_lock here because we have no promises about
>> + * concurrent writes anyway.
>> + */
>> +static loff_t seek_extend_unmapped_end(struct qio *qio, struct 
>> qcow2_map *map,
>> +                       loff_t end)
>> +{
>> +    struct qcow2 *qcow2 = qio->qcow2;
>> +    loff_t lim = SEEK_QIO_DATA(qio)->lim;
>> +    u32 step = 1 + qcow2->ext_l2;
>> +    u32 index = map->l2.index_in_page + step;
>> +    u64 *entries;
>> +
>> +    entries = kmap_local_page(map->l2.md->page);
>> +    for (; index + step <= PAGE_SIZE / sizeof(u64) && end < lim;
>> +         index += step) {
>> +        /* Zero test does not need be64_to_cpu() */
>> +        if (READ_ONCE(entries[index]) ||
>> +            (qcow2->ext_l2 && READ_ONCE(entries[index + 1])))
>> +            break;
>> +        end += qcow2->clu_size;
>> +    }
>> +    kunmap_local(entries);
>> +
>> +    return min_t(loff_t, end, lim);
>> +}
>> +
>>   static struct qio *advance_and_spawn_lower_seek_qio(struct qio 
>> *old_qio, loff_t end)
>>   {
>>       struct qcow2 *lower = old_qio->qcow2->lower;
>> @@ -4503,12 +4531,15 @@ static int qcow2_llseek_hole_qio(struct qio 
>> *qio, int whence, loff_t *result)
>>                   struct qio *new_qio;
>>                   loff_t end;
>> -                if (!(map.level & L2_LEVEL))
>> +                if (!(map.level & L2_LEVEL)) {
>>                       end = min_t(loff_t,
>>                               to_bytes(get_next_l2(qio)),
>>                               SEEK_QIO_DATA(qio)->lim);
>> -                else
>> +                } else {
>>                       end = to_bytes(qio->bi_iter.bi_sector) + size;
>> +                    if (!CLU_OFF(qio->qcow2, end))
>> +                        end = seek_extend_unmapped_end(qio, &map, end);
>> +                }
>>                   new_qio = advance_and_spawn_lower_seek_qio(qio, end);
>>                   if (!new_qio) {
>> @@ -4552,6 +4583,15 @@ static int qcow2_llseek_hole_qio(struct qio 
>> *qio, int whence, loff_t *result)
>>                   qio->bi_iter.bi_sector += to_sector(size);
>>                   qio->bi_iter.bi_size -= size;
>>                   goto calc_subclu;
>> +            } else if (arg.unmapped && (map.level & L2_LEVEL)) {
>> +                loff_t end = to_bytes(qio->bi_iter.bi_sector) + size;
>> +
>> +                /* Skip the following unallocated entries in one go */
>> +                if (!CLU_OFF(qio->qcow2, end)) {
>> +                    seek_qio_set_sector(qio,
> 
> shouldn't you wrap the result of seek_extend_unmapped_end in to_sector? 
> It returns bytes and seek_qio_set_sector takes sector_t and we would 
> immediately do to_bytes on something that is already bytes.

You are totally right. No idea how I missed that! Probably tests I 
drafted into ploop do not cover that case.

> 
>> +                        seek_extend_unmapped_end(qio, &map, end));
>> +                    continue;
>> +                }
>>               }
>>           }
> 



More information about the Devel mailing list