[Devel] [PATCH VZ10 v3 3/9] ve/fs: Rework per-ve mount count

Vasileios Almpanis vasileios.almpanis at virtuozzo.com
Mon Jul 27 10:35:13 MSK 2026


On 7/26/26 11:30 PM, Vladimir Riabchun wrote:
> Previous approach with current mounts counter had an issue:
> there was a gap between ve_mount_allowed check and ve_mount_nr_inc,
> which could allow CT to have more mounts than expected.
>
> Fix this by tracking the number of available mounts instead
> of current ones. This also makes resources accounting
> more consistent - we are using ***_avail_nr approach more.
>
> One more issue with inconsistent ve value is fixed:
> ve_mount_allowed always used ve from get_exec_env, but
> ve_mount_nr_inc operated with owner_ve.
> Now actual ve value is calculated in the beginning of alloc_vfsmnt.
>
> We can't ignore pseudouser mode, otherwise CRIU may fail when
> the number of mounts is close to the limit.
> It is possible to create a mount as a pseudouser and then do umount
> as a normal user. To prevent incorrect accounting and treat
> pseudouser mode special, a new field created_as_pseudouser was added
> to mount structure.
>
> https://virtuozzo.atlassian.net/browse/VSTOR-135520
>
> Feature: per-ve failcounters
> Signed-off-by: Vladimir Riabchun <vladimir.riabchun at virtuozzo.com>
> ---
>   fs/mount.h         |  1 +
>   fs/namespace.c     | 66 ++++++++++++++++++++++++++--------------------
>   include/linux/ve.h |  2 +-
>   kernel/ve/ve.c     | 12 ++++-----
>   4 files changed, 45 insertions(+), 36 deletions(-)
>
> diff --git a/fs/mount.h b/fs/mount.h
> index 5cf06431d586..2ea5248c64e0 100644
> --- a/fs/mount.h
> +++ b/fs/mount.h
> @@ -72,6 +72,7 @@ struct mount {
>   	struct list_head mnt_umounting; /* list entry for umount propagation */
>   #ifdef CONFIG_VE
>   	struct ve_struct *ve_owner;	/* VE in which this mount was created */
> +	bool created_as_pseudouser;	/* pseudouser mounts ignore per-VE limits */
>   #endif /* CONFIG_VE */
>   #ifdef CONFIG_FSNOTIFY
>   	struct fsnotify_mark_connector __rcu *mnt_fsnotify_marks;
> diff --git a/fs/namespace.c b/fs/namespace.c
> index cd6aa2127203..6811082360cc 100644
> --- a/fs/namespace.c
> +++ b/fs/namespace.c
> @@ -317,18 +317,21 @@ int mnt_get_count(struct mount *mnt)
>   #endif
>   }
>   
> -static inline int ve_mount_allowed(void);
> -static inline void ve_mount_nr_inc(struct mount *mnt, struct ve_struct *ve);
> -static inline void ve_mount_nr_dec(struct mount *mnt);
> +static inline int ve_try_reserve_mount(struct ve_struct *ve);
> +static inline void ve_mount_put(struct mount *mnt, struct ve_struct *ve);
>   
>   static struct mount *alloc_vfsmnt(const char *name, struct ve_struct *owner_ve)
>   {
>   	struct mount *mnt;
> +	struct ve_struct *ve = owner_ve;
>   
> -	if (!ve_mount_allowed()) {
> +	if (!ve)
> +		ve = get_exec_env();
> +
> +	if (!ve_try_reserve_mount(ve)) {
>   		pr_warn_ratelimited(
>   			"CT#%s reached the limit on mounts.\n",
> -			ve_name(get_exec_env()));
> +			ve_name(ve));
>   		return NULL;
>   	}
>   
> @@ -370,7 +373,13 @@ static struct mount *alloc_vfsmnt(const char *name, struct ve_struct *owner_ve)
>   		INIT_LIST_HEAD(&mnt->mnt_umounting);
>   		INIT_HLIST_HEAD(&mnt->mnt_stuck_children);
>   		mnt->mnt.mnt_idmap = &nop_mnt_idmap;
> -		ve_mount_nr_inc(mnt, owner_ve);
> +#ifdef CONFIG_VE
> +		/* Got ve reference in ve_try_reserve_mount */
> +		mnt->ve_owner = ve;
> +		mnt->created_as_pseudouser = ve->is_pseudosuper;
> +#endif
> +	} else {
> +		goto out_nomem;
>   	}
>   	return mnt;
>   
> @@ -382,6 +391,9 @@ static struct mount *alloc_vfsmnt(const char *name, struct ve_struct *owner_ve)
>   	mnt_free_id(mnt);
>   out_free_cache:
>   	kmem_cache_free(mnt_cache, mnt);
Isn't this use-after-free? If for some reason we go to either 
out_free_devname, out_free_id or out_free_cache, mnt is freed but 
remains != NULL and is passed to ve_mount_put where we read 
mnt->created_as_pseudouser, and also write mnt->ve_owner = NULL. Here 
since we will reserve and put in the same place maybe you could add some 
helper that just checks if ve is super or pseudosuper and increments the 
number of available mounts, or just pass NULL to ve_mount_put
> +out_nomem:
> +	/* Got ve reference in ve_try_reserve_mount */
> +	ve_mount_put(mnt, ve);
>   	return NULL;
>   }
>   
> @@ -750,7 +762,7 @@ int sb_prepare_remount_readonly(struct super_block *sb)
>   static void free_vfsmnt(struct mount *mnt)
>   {
>   	mnt_idmap_put(mnt_idmap(&mnt->mnt));
> -	ve_mount_nr_dec(mnt);
> +	ve_mount_put(mnt, mnt->ve_owner);
>   	kfree_const(mnt->mnt_devname);
>   #ifdef CONFIG_SMP
>   	free_percpu(mnt->mnt_pcp);
> @@ -3258,30 +3270,27 @@ int ve_devmnt_process(struct ve_struct *ve, dev_t dev, void **data_pp, int remou
>   	return err;
>   }
>   
> -static inline int ve_mount_allowed(void)
> -{
> -	struct ve_struct *ve = get_exec_env();
> -
> -	return ve_is_super(ve) || ve->is_pseudosuper ||
> -		atomic_read(&ve->mnt_nr) < (int)sysctl_ve_mount_nr;
> -}
> -
> -static inline void ve_mount_nr_inc(struct mount *mnt, struct ve_struct *ve)
> +static inline int ve_try_reserve_mount(struct ve_struct *ve)
>   {
> -	if (!ve)
> -		ve = get_exec_env();
> -
> -	mnt->ve_owner = get_ve(ve);
> -	atomic_inc(&ve->mnt_nr);
> +	int ret = ve_is_super(ve) || ve->is_pseudosuper ||
> +		  atomic_dec_if_positive(&ve->mnt_avail_nr) >= 0;
> +	if (ret)
> +		get_ve(ve);
> +	return ret;
>   }
>   
> -static inline void ve_mount_nr_dec(struct mount *mnt)
> +static inline void ve_mount_put(struct mount *mnt, struct ve_struct *ve)
>   {
> -	struct ve_struct *ve = mnt->ve_owner;
> -
> -	atomic_dec(&ve->mnt_nr);
> +	/*
> +	 * ve argument is needed to reuse this function in alloc_vfsmnt error path.
> +	 * Other users should pass mnt->ve_owner value.
> +	 */
> +	if (!ve_is_super(ve) && !(mnt && mnt->created_as_pseudouser) &&
> +	    !(!mnt && ve->is_pseudosuper))
> +		atomic_inc(&ve->mnt_avail_nr);
>   	put_ve(ve);
> -	mnt->ve_owner = NULL;
> +	if (mnt)
> +		mnt->ve_owner = NULL;
>   }
>   
>   bool is_sb_ve_accessible(struct ve_struct *ve, struct super_block *sb)
> @@ -3303,9 +3312,8 @@ bool is_sb_ve_accessible(struct ve_struct *ve, struct super_block *sb)
>   
>   #else /* CONFIG_VE */
>   
> -static inline int ve_mount_allowed(void) { return 1; }
> -static inline void ve_mount_nr_inc(struct mount *mnt, struct ve_struct *ve) { }
> -static inline void ve_mount_nr_dec(struct mount *mnt) { }
> +static inline int ve_try_reserve_mount(struct ve_struct *ve) { return 1; }
> +static inline void ve_mount_put(struct mount *mnt, struct ve_struct *ve) { }
>   #endif /* CONFIG_VE */
>   
>   static int ve_prepare_mount_options(struct fs_context *fc, void *data)
> diff --git a/include/linux/ve.h b/include/linux/ve.h
> index 3facbd1759df..cca0a2bc1aac 100644
> --- a/include/linux/ve.h
> +++ b/include/linux/ve.h
> @@ -88,7 +88,7 @@ struct ve_struct {
>   	atomic_t		nd_neigh_nr;
>   	unsigned long		meminfo_val;
>   
> -	atomic_t		mnt_nr; /* number of present VE mounts */
> +	atomic_t		mnt_avail_nr; /* number of available VE mounts */
>   
>   #ifdef CONFIG_COREDUMP
>   	char			core_pattern[CORENAME_MAX_SIZE];
> diff --git a/kernel/ve/ve.c b/kernel/ve/ve.c
> index dddf2393326d..3f66144eeb7e 100644
> --- a/kernel/ve/ve.c
> +++ b/kernel/ve/ve.c
> @@ -81,7 +81,7 @@ struct ve_struct ve0 = {
>   
>   	.arp_neigh_nr		= ATOMIC_INIT(0),
>   	.nd_neigh_nr		= ATOMIC_INIT(0),
> -	.mnt_nr			= ATOMIC_INIT(0),
> +	.mnt_avail_nr		= ATOMIC_INIT(INT_MAX),
>   	.meminfo_val		= VE_MEMINFO_SYSTEM,
>   	.umh_running_helpers	= ATOMIC_INIT(0),
>   	.umh_helpers_waitq	= __WAIT_QUEUE_HEAD_INITIALIZER(ve0.umh_helpers_waitq),
> @@ -778,7 +778,7 @@ static struct cgroup_subsys_state *ve_create(struct cgroup_subsys_state *parent_
>   
>   	atomic_set(&ve->arp_neigh_nr, 0);
>   	atomic_set(&ve->nd_neigh_nr, 0);
> -	atomic_set(&ve->mnt_nr, 0);
> +	atomic_set(&ve->mnt_avail_nr, sysctl_ve_mount_nr);
>   
>   #ifdef CONFIG_COREDUMP
>   	strcpy(ve->core_pattern, "core");
> @@ -1055,9 +1055,9 @@ static u64 ve_netns_avail_nr_read(struct cgroup_subsys_state *css, struct cftype
>   	return atomic_read(&css_to_ve(css)->netns_avail_nr);
>   }
>   
> -static u64 ve_mnt_nr_read(struct cgroup_subsys_state *css, struct cftype *cft)
> +static s64 ve_mnt_avail_nr_read(struct cgroup_subsys_state *css, struct cftype *cft)
>   {
> -	return atomic_read(&css_to_ve(css)->mnt_nr);
> +	return atomic_read(&css_to_ve(css)->mnt_avail_nr);
>   }
>   
>   static u64 ve_netif_max_nr_read(struct cgroup_subsys_state *css, struct cftype *cft)
> @@ -1617,8 +1617,8 @@ static struct cftype ve_cftypes[] = {
>   		.read_u64		= ve_netns_avail_nr_read,
>   	},
>   	{
> -		.name			= "mnt_nr",
> -		.read_u64		= ve_mnt_nr_read,
> +		.name			= "mnt_avail_nr",
> +		.read_s64		= ve_mnt_avail_nr_read,
>   	},
>   	{
>   		.name			= "netif_max_nr",

-- 
Best regards, Vasileios Almpanis
Software Developer, Virtuozzo.



More information about the Devel mailing list