一次挂载的完整旅程:Linux 内核 VFS 到 ext4 的调用链剖析

挂载(mount)是文件系统与用户空间之间的第一座桥梁,也是内核中最容易出问题、最难调试的路径之一。Linux 内核在近几个版本中对挂载 API 做了大规模重构,引入了以 fs_context 为中心的新挂载框架,把"参数解析"与"超级块获取"从传统 mount() 系统调用的泥潭里剥离出来。本文以 Linux 6.18 源码为蓝本,沿着 do_new_mount() → fs_context_for_mount() → vfs_get_tree() → ext4_get_tree() → get_tree_bdev() → sget_fc() → ext4_fill_super() → __ext4_fill_super() 这条主链,逐层剖析一次挂载请求是如何从用户空间落地到 ext4 超级块的。

一、入口层:do_new_mount 的四个阶段

一切从 do_new_mount() 开始。它是传统 mount() 系统调用进入 VFS 后的核心函数之一,也是新旧挂载 API 的交汇点:

c

复制代码
static int do_new_mount(struct path *path, const char *fstype, int sb_flags,
			int mnt_flags, const char *name, void *data)
{
	struct file_system_type *type;
	struct fs_context *fc;
	const char *subtype = NULL;
	int err = 0;

	if (!fstype)
		return -EINVAL;

	type = get_fs_type(fstype);          // 【阶段 1】查找文件系统类型
	if (!type)
		return -ENODEV;

	if (type->fs_flags & FS_HAS_SUBTYPE) {
		subtype = strchr(fstype, '.');   // 处理 "ext4.dev" 这类子类型
		if (subtype) {
			subtype++;
			if (!*subtype) {
				put_filesystem(type);
				return -EINVAL;
			}
		}
	}

	fc = fs_context_for_mount(type, sb_flags);  // 【阶段 2】创建上下文
	put_filesystem(type);
	if (IS_ERR(fc))
		return PTR_ERR(fc);

	/*
	 * Indicate to the filesystem that the mount request is coming
	 * from the legacy mount system call.
	 */
	fc->oldapi = true;                   // 标记来自传统 mount() 调用

	if (subtype)
		err = vfs_parse_fs_string(fc, "subtype",
					  subtype, strlen(subtype));
	if (!err && name)
		err = vfs_parse_fs_string(fc, "source", name, strlen(name));
	if (!err)
		err = parse_monolithic_mount_data(fc, data);  // 【阶段 3】解析参数
	if (!err && !mount_capable(fc))
		err = -EPERM;
	if (!err)
		err = vfs_get_tree(fc);          // 【阶段 4a】获取超级块
	if (!err)
		err = do_new_mount_fc(fc, path, mnt_flags);  // 【阶段 4b】加入命名空间

	put_fs_context(fc);
	return err;
}

这段代码的结构非常清晰,可以概括为四个阶段:

  1. 类型查找 :get_fs_type() 根据用户传入的字符串(如 "ext4")找到对应的 struct file_system_type。

  2. 上下文创建 :fs_context_for_mount() 分配并初始化 struct fs_context,绑定文件系统的操作集。

  3. 参数解析 :通过 vfs_parse_fs_string() 与 parse_monolithic_mount_data() 把用户参数填入上下文。

  4. 超级块获取与挂载 :vfs_get_tree() 触发文件系统去获取或创建超级块,do_new_mount_fc() 把挂载点加入当前命名空间。

值得注意的是,fc->oldapi = true 这一行标记了上下文来自旧 API,文件系统可以据此决定是否走兼容路径。这正是新框架"新旧兼容"的设计精髓------fs_context 作为统一的配置载体,既支持新式的"逐参数解析",也支持传统的一整块 data 字符串。

二、类型查找:get_fs_type 与模块自动加载

挂载的第一步是找到"我要挂哪种文件系统"。这个任务由 get_fs_type() 承担:

c

复制代码
struct file_system_type *get_fs_type(const char *name)
{
	struct file_system_type *fs;
	const char *dot = strchr(name, '.');
	int len = dot ? dot - name : strlen(name);

	fs = __get_fs_type(name, len);       // 先在已注册链表里找
	if (!fs && (request_module("fs-%.*s", len, name) == 0)) {
		fs = __get_fs_type(name, len);   // 尝试按需加载模块后再找
		if (!fs)
			pr_warn_once("request_module fs-%.*s succeeded, but still no fs?\n",
				     len, name);
	}

	if (dot && fs && !(fs->fs_flags & FS_HAS_SUBTYPE)) {
		put_filesystem(fs);              // 类型不支持子类型则拒绝
		fs = NULL;
	}
	return fs;
}

这里有一个容易被忽略的细节:request_module() 会在用户态调用 modprobe,尝试加载名为 fs-ext4 之类的模块。如果模块加载成功但仍未注册文件系统,内核会打印一次警告。这个"先查找、后加载、再查找"的双阶段逻辑保证了按需加载与并发安全。

__get_fs_type() 在 file_systems_lock 保护下,通过 find_filesystem() 遍历全局 file_systems 链表,并用 try_module_get() 增加模块引用计数:

c

复制代码
static struct file_system_type *__get_fs_type(const char *name, int len)
{
	struct file_system_type *fs;

	read_lock(&file_systems_lock);
	fs = *(find_filesystem(name, len));
	if (fs && !try_module_get(fs->owner))
		fs = NULL;
	read_unlock(&file_systems_lock);
	return fs;
}

try_module_get() 是关键------它保证了文件系统模块在挂载过程中不会被卸载。对应的 put_filesystem() 在 do_new_mount() 里会在 fs_context_for_mount() 之后调用,意味着文件系统的生命周期由 fs_context 接管。

三、上下文创建:fs_context_for_mount 与 alloc_fs_context

拿到 file_system_type 后,do_new_mount() 调用 fs_context_for_mount(),后者只是 alloc_fs_context() 的薄封装:

c

复制代码
struct fs_context *fs_context_for_mount(struct file_system_type *fs_type,
					unsigned int sb_flags)
{
	return alloc_fs_context(fs_type, NULL, sb_flags, 0,
					FS_CONTEXT_FOR_MOUNT);
}

alloc_fs_context() 是核心,它承担了上下文初始化的大部分工作:

c

复制代码
static struct fs_context *alloc_fs_context(struct file_system_type *fs_type,
				      struct dentry *reference,
				      unsigned int sb_flags,
				      unsigned int sb_flags_mask,
				      enum fs_context_purpose purpose)
{
	int (*init_fs_context)(struct fs_context *);
	struct fs_context *fc;
	int ret = -ENOMEM;

	fc = kzalloc(sizeof(struct fs_context), GFP_KERNEL_ACCOUNT);
	if (!fc)
		return ERR_PTR(-ENOMEM);

	fc->purpose	= purpose;
	fc->sb_flags	= sb_flags;
	fc->sb_flags_mask = sb_flags_mask;
	fc->fs_type	= get_filesystem(fs_type);
	fc->cred	= get_current_cred();          // 保存当前进程的 cred
	fc->net_ns	= get_net(current->nsproxy->net_ns);
	fc->log.prefix	= fs_type->name;

	mutex_init(&fc->uapi_mutex);

	switch (purpose) {
	case FS_CONTEXT_FOR_MOUNT:
		fc->user_ns = get_user_ns(fc->cred->user_ns);  // 使用当前用户命名空间
		break;
	case FS_CONTEXT_FOR_SUBMOUNT:
		fc->user_ns = get_user_ns(reference->d_sb->s_user_ns);
		break;
	case FS_CONTEXT_FOR_RECONFIGURE:
		atomic_inc(&reference->d_sb->s_active);
		fc->user_ns = get_user_ns(reference->d_sb->s_user_ns);
		fc->root = dget(reference);
		break;
	}

	/* TODO: Make all filesystems support this unconditionally */
	init_fs_context = fc->fs_type->init_fs_context;
	if (!init_fs_context)
		init_fs_context = legacy_init_fs_context;  // 旧文件系统走兼容路径

	ret = init_fs_context(fc);
	if (ret < 0)
		goto err_fc;
	fc->need_free = true;
	return fc;

err_fc:
	put_fs_context(fc);
	return ERR_PTR(ret);
}

这段代码体现了新挂载框架的三个设计要点:

第一,上下文与用途绑定。 purpose 有三种取值------FOR_MOUNT(新挂载)、FOR_SUBMOUNT(子挂载/自动挂载)、FOR_RECONFIGURE(重新配置)。不同用途决定了 user_ns 的取值来源:新挂载使用当前进程的用户命名空间,子挂载与重配置则继承超级块的用户命名空间。这直接影响后续的权限检查。

第二,init_fs_context 回调是扩展点。 新式文件系统实现了自己的 init_fs_context,可以在此时分配私有数据、绑定操作集。旧文件系统没有这个回调,内核回退到 legacy_init_fs_context(),由后者分配一个兼容上下文。

第三,资源引用在上下文里集中管理。 get_filesystem()、get_current_cred()、get_net()、get_user_ns() 都在这里获取,对应的 put_fs_context() 统一释放,避免了"半路出错忘记释放"的经典问题。

对 ext4 来说,init_fs_context 指向 ext4_init_fs_context():

c

复制代码
int ext4_init_fs_context(struct fs_context *fc)
{
	struct ext4_fs_context *ctx;

	ctx = kzalloc(sizeof(struct ext4_fs_context), GFP_KERNEL);
	if (!ctx)
		return -ENOMEM;

	fc->fs_private = ctx;                // 私有数据:ext4 参数解析状态
	fc->ops = &ext4_context_ops;         // 绑定 ext4 的操作集

	return 0;
}

ext4_context_ops 定义了 ext4 在挂载过程中的四个关键回调:

c

复制代码
static const struct fs_context_operations ext4_context_ops = {
	.parse_param	= ext4_parse_param,   // 解析单个挂载参数
	.get_tree	= ext4_get_tree,      // 获取超级块根
	.reconfigure	= ext4_reconfigure,   // 重新配置
	.free		= ext4_fc_free,       // 释放私有数据
};

至此,fs_context 已经完全就绪:它知道"要挂什么类型(fs_type)"、"用什么权限(cred、user_ns)"、"由谁处理(ops)"。

四、获取挂载树:vfs_get_tree 的承上启下

vfs_get_tree() 是 VFS 层获取超级块根的核心函数,也是连接通用框架与文件系统实现的桥梁:

c

复制代码
int vfs_get_tree(struct fs_context *fc)
{
	struct super_block *sb;
	int error;

	if (fc->root)
		return -EBUSY;                   // 已经获取过,拒绝重复

	/* Get the mountable root in fc->root, with a ref on the root and a ref
	 * on the superblock.
	 */
	error = fc->ops->get_tree(fc);       // 【关键】调用文件系统特定实现
	if (error < 0)
		return error;

	if (!fc->root) {
		pr_err("Filesystem %s get_tree() didn't set fc->root\n",
		       fc->fs_type->name);
		BUG();                           // 契约:get_tree 必须设置 fc->root
	}

	sb = fc->root->d_sb;
	WARN_ON(!sb->s_bdi);

	/*
	 * super_wake() contains a memory barrier which also care of
	 * ordering for super_cache_count(). We place it before setting
	 * SB_BORN as the data dependency between the two functions is
	 * the superblock structure contents that we just set up, not
	 * the SB_BORN flag.
	 */
	super_wake(sb, SB_BORN);             // 唤醒超级块,标记 SB_BORN

	error = security_sb_set_mnt_opts(sb, fc->security, 0, NULL);
	if (unlikely(error)) {
		fc_drop_locked(fc);
		return error;
	}

	WARN((sb->s_maxbytes < 0), "%s set sb->s_maxbytes to "
		"negative value (%lld)\n", fc->fs_type->name, sb->s_maxbytes);

	return 0;
}

这段代码有几个值得关注的细节:

契约式的 BUG() :如果文件系统实现了 get_tree 却没有设置 fc->root,内核直接 BUG()。这不是"防御性编程",而是接口契约的强制执行 ------get_tree 的唯一职责就是设置根 dentry,做不到就是文件系统的错,没有回退余地。

super_wake(sb, SB_BORN) 与内存屏障 :注释里专门解释了为什么要放在 SB_BORN 之前------super_wake() 内含内存屏障,它保护的是 super_cache_count() 的顺序。由于数据依赖是超级块结构体的内容(刚设置好),而不是 SB_BORN 标志本身,所以顺序必须是这样。这种"注释即文档"的风格在内核里很常见,也是阅读内核源码时最值得留意的部分。

security_sb_set_mnt_opts() :LSM(如 SELinux)在此刻介入,设置挂载点的安全选项。如果 LSM 拒绝,fc_drop_locked() 会释放已获取的根与超级块,整个挂载优雅失败。

五、ext4 的实现:ext4_get_tree 与 get_tree_bdev

vfs_get_tree() 通过 fc->ops->get_tree(fc) 回调到 ext4 的 ext4_get_tree():

c

复制代码
static int ext4_get_tree(struct fs_context *fc)
{
	return get_tree_bdev(fc, ext4_fill_super);  // 委托给块设备通用辅助
}

ext4 作为块设备文件系统,把大部分逻辑委托给了 get_tree_bdev()。这个函数是所有基于块设备的文件系统的公共辅助:

c

复制代码
int get_tree_bdev(struct fs_context *fc,
		int (*fill_super)(struct super_block *,
				  struct fs_context *))
{
	struct super_block *s;
	int error = 0;
	dev_t dev;

	if (!fc->source)
		return invalf(fc, "No source specified");

	error = lookup_bdev(fc->source, &dev);   // 设备路径 → dev_t
	if (error) {
		errorf(fc, "%s: Can't lookup blockdev", fc->source);
		return error;
	}

	fc->sb_flags |= SB_NOSEC;
	s = sget_dev(fc, dev);                   // 查找或创建超级块
	if (IS_ERR(s))
		return PTR_ERR(s);

	if (s->s_root) {
		/* Don't summarily change the RO/RW state. */
		if ((fc->sb_flags ^ s->s_flags) & SB_RDONLY) {
			warnf(fc, "%pg: Can't mount, would change RO state", s->s_bdev);
			deactivate_locked_super(s);
			return -EBUSY;
		}
	} else {
		error = setup_bdev_super(s, fc->sb_flags, fc);
		if (!error)
			error = fill_super(s, fc);   // 【关键】调用 ext4_fill_super
		if (error) {
			deactivate_locked_super(s);
			return error;
		}
		s->s_flags |= SB_ACTIVE;
	}

	BUG_ON(fc->root);
	fc->root = dget(s->s_root);              // 设置可挂载根
	return 0;
}

这段代码的逻辑分支非常关键,涉及"超级块是否已存在"两种情况:

情况一:超级块已存在(s->s_root 非空)。 说明该设备已被挂载过。此时不能简单地重新走一遍 fill_super,只需要检查只读状态是否冲突。(fc->sb_flags ^ s->s_flags) & SB_RDONLY 这个异或运算很巧妙------它只关心 SB_RDONLY 这一位是否不同,如果不同(比如原来是 r/o 现在想挂 r/w),则拒绝并返回 -EBUSY。这防止了"先 ro 后 rw 挂同一个设备"造成的状态不一致。

情况二:超级块是新建的。 依次执行 setup_bdev_super()(设置 s_bdev、s_bdi 等基本字段)和 fill_super()(即 ext4_fill_super),成功后设置 SB_ACTIVE。

无论哪种情况,最后都通过 fc->root = dget(s->s_root) 把根 dentry 交给上下文。这个 BUG_ON(fc->root) 是防御性检查------如果之前的流程已经设置过根,说明有代码路径重复调用,属于逻辑错误。

六、超级块的查找与创建:sget_fc 的并发控制

get_tree_bdev() 里的 sget_dev() 最终调用到 sget_fc():

c

复制代码
struct super_block *sget_dev(struct fs_context *fc, dev_t dev)
{
	fc->sget_key = &dev;
	return sget_fc(fc, super_s_dev_test, super_s_dev_set);
}

sget_fc() 是超级块生命周期管理的核心,它要解决的核心问题是:多个进程同时挂载同一设备时,如何保证只创建一个超级块?

c

复制代码
struct super_block *sget_fc(struct fs_context *fc,
			    int (*test)(struct super_block *, struct fs_context *),
			    int (*set)(struct super_block *, struct fs_context *))
{
	struct super_block *s = NULL;
	struct super_block *old;
	struct user_namespace *user_ns = fc->global ? &init_user_ns : fc->user_ns;
	int err;

retry:
	spin_lock(&sb_lock);
	if (test) {
		hlist_for_each_entry(old, &fc->fs_type->fs_supers, s_instances) {
			if (test(old, fc))
				goto share_extant_sb;
		}
	}
	if (!s) {
		spin_unlock(&sb_lock);
		s = alloc_super(fc->fs_type, fc->sb_flags, user_ns);
		if (!s)
			return ERR_PTR(-ENOMEM);
		goto retry;                      // 分配后重试,重新查找
	}

	s->s_fs_info = fc->s_fs_info;
	err = set(s, fc);
	if (err) {
		s->s_fs_info = NULL;
		spin_unlock(&sb_lock);
		destroy_unused_super(s);
		return ERR_PTR(err);
	}
	fc->s_fs_info = NULL;
	s->s_type = fc->fs_type;
	s->s_iflags |= fc->s_iflags;
	strscpy(s->s_id, s->s_type->name, sizeof(s->s_id));
	/*
	 * Make the superblock visible on @super_blocks and @fs_supers.
	 * It's in a nascent state and users should wait on SB_BORN or
	 * SB_DYING to be set.
	 */
	list_add_tail(&s->s_list, &super_blocks);
	hlist_add_head(&s->s_instances, &s->s_type->fs_supers);
	spin_unlock(&sb_lock);
	get_filesystem(s->s_type);
	shrinker_register(s->s_shrink);
	return s;

share_extant_sb:
	if (user_ns != old->s_user_ns || fc->exclusive) {
		spin_unlock(&sb_lock);
		destroy_unused_super(s);
		if (fc->exclusive)
			warnfc(fc, "reusing existing filesystem not allowed");
		else
			warnfc(fc, "reusing existing filesystem in another namespace not allowed");
		return ERR_PTR(-EBUSY);
	}
	if (!grab_super(old))
		goto retry;
	destroy_unused_super(s);
	return old;
}

这段代码有几个精妙之处:

重试机制(goto retry) :sb_lock 是自旋锁,不能睡眠。所以 alloc_super() 必须在释放锁之后调用。但释放锁后,其它 CPU 可能已经创建了同样的超级块,所以分配完必须重新加锁、重新查找。这种"先释放锁分配,再重试查找"的模式是内核里处理"锁内不能睡眠"场景的经典范式。

test 与 set 回调 :test 用于判断两个超级块是否"相同"(对块设备文件系统而言是比较设备号),set 用于在新建超级块时写入关键字段(比如 s_dev)。这两个回调抽象了不同文件系统的"超级块身份判定"逻辑。

命名空间与独占检查 :share_extant_sb 标签下的逻辑很关键------如果现有超级块的 s_user_ns 与当前上下文的 user_ns 不同,或者上下文要求 exclusive(独占,不允许复用),就返回 -EBUSY。这是容器时代挂载隔离的核心机制------不允许跨用户命名空间共享超级块,否则容器 A 就能通过重新挂载看到容器 B 的文件系统。

状态可见性 :新建的超级块通过 list_add_tail() 和 hlist_add_head() 分别加入 super_blocks 全局链表与 fs_type->fs_supers 哈希链。注释里强调它处于"nascent(初生)状态",其它代码应当等待 SB_BORN 或 SB_DYING。这解释了为什么 vfs_get_tree() 要在 get_tree 成功后才调用 super_wake(sb, SB_BORN)------顺序不能颠倒。

七、ext4 的超级块填充:从 ext4_fill_super 到 __ext4_fill_super

超级块新建后,get_tree_bdev() 调用 fill_super(),即 ext4_fill_super():

c

复制代码
static int ext4_fill_super(struct super_block *sb, struct fs_context *fc)
{
	struct ext4_fs_context *ctx = fc->fs_private;
	struct ext4_sb_info *sbi;
	const char *descr;
	int ret;

	sbi = ext4_alloc_sbi(sb);                // 分配 ext4 私有超级块信息
	if (!sbi)
		return -ENOMEM;

	fc->s_fs_info = sbi;

	strreplace(sb->s_id, '/', '!');          // 清理名字中的 '/'

	sbi->s_sb_block = 1;	/* Default super block location */
	if (ctx->spec & EXT4_SPEC_s_sb_block)
		sbi->s_sb_block = ctx->s_sb_block;

	ret = __ext4_fill_super(fc, sb);         // 【核心】实际的填充逻辑
	//yym-gaizao
	pr_info("YYM-DBG: [ext4_fill_super] Attempting to fill ext4 superblock for device %s \n", sb->s_id);
	if (ret < 0)
		goto free_sbi;

	if (sbi->s_journal) {
		if (test_opt(sb, DATA_FLAGS) == EXT4_MOUNT_JOURNAL_DATA)
			descr = " journalled data mode";
		else if (test_opt(sb, DATA_FLAGS) == EXT4_MOUNT_ORDERED_DATA)
			descr = " ordered data mode";
		else
			descr = " writeback data mode";
	} else
		descr = "out journal";

	if (___ratelimit(&ext4_mount_msg_ratelimit, "EXT4-fs mount"))
		ext4_msg(sb, KERN_INFO, "mounted filesystem %pU %s with%s. "
			 "Quota mode: %s.", &sb->s_uuid,
			 sb_rdonly(sb) ? "ro" : "r/w", descr,
			 ext4_quota_mode(sb));

	/* Update the s_overhead_clusters if necessary */
	ext4_update_overhead(sb, false);
	return 0;

free_sbi:
	ext4_free_sbi(sbi);
	fc->s_fs_info = NULL;
	return ret;
}

ext4_fill_super() 本身只做"外壳"工作------分配 sbi、绑定到上下文、调用 __ext4_fill_super()、打印挂载信息。真正的重头戏在 __ext4_fill_super() 里,它有超过 400 行代码,是 ext4 挂载中最复杂、最关键的部分。挑出主线步骤:

c

复制代码
static int __ext4_fill_super(struct fs_context *fc, struct super_block *sb)
{
	...
	/* 1. 读取磁盘超级块 */
	err = ext4_load_super(sb, &logical_sb_block, silent);
	if (err)
		goto out_fail;

	/* 2. 初始化元数据校验和 */
	err = ext4_init_metadata_csum(sb, es);
	if (err)
		goto failed_mount;

	/* 3. 设置默认挂载选项 */
	ext4_set_def_opts(sb, es);

	/* 4. 初始化 inode 相关信息 */
	err = ext4_inode_info_init(sb, es);
	if (err)
		goto failed_mount;

	/* 5. 应用超级块中的挂载选项 */
	err = parse_apply_sb_mount_options(sb, ctx);
	if (err < 0)
		goto failed_mount;

	/* 6. 检查选项一致性并应用 */
	err = ext4_check_opt_consistency(fc, sb);
	if (err < 0)
		goto failed_mount;
	ext4_apply_options(fc, sb);

	/* 7. 初始化编码、检查特性兼容性 */
	err = ext4_encoding_init(sb, es);
	if (err)
		goto failed_mount;
	err = ext4_check_feature_compatibility(sb, es, silent);
	if (err)
		goto failed_mount;

	/* 8. 初始化块组元数据、几何信息 */
	err = ext4_block_group_meta_init(sb, silent);
	if (err)
		goto failed_mount;
	err = ext4_group_desc_init(sb, es, logical_sb_block, &first_not_zeroed);
	if (err)
		goto failed_mount3;

	/* 9. 注册 extent 状态收缩器 */
	err = ext4_es_register_shrinker(sbi);
	if (err)
		goto failed_mount3;

	/* 10. 加载并初始化日志 */
	if (!test_opt(sb, NOLOAD) && ext4_has_feature_journal(sb)) {
		err = ext4_load_and_init_journal(sb, es, ctx);
		if (err)
			goto failed_mount3a;
	}
	...

	/* 11. 获取根 inode,创建根 dentry */
	root = ext4_iget(sb, EXT4_ROOT_INO, EXT4_IGET_SPECIAL);
	if (IS_ERR(root)) {
		ext4_msg(sb, KERN_ERR, "get root inode failed");
		err = PTR_ERR(root);
		root = NULL;
		goto failed_mount4;
	}
	sb->s_root = d_make_root(root);          // 【关键】建立超级块根
	if (!sb->s_root) {
		ext4_msg(sb, KERN_ERR, "get root dentry failed");
		err = -ENOMEM;
		goto failed_mount4;
	}

	/* 12. 初始化 mballoc(多块分配器) */
	err = ext4_mb_init(sb);
	if (err) {
		ext4_msg(sb, KERN_ERR, "failed to initialize mballoc (%d)", err);
		goto failed_mount5;
	}

	/* 13. 注册 sysfs、初始化孤儿信息 */
	err = ext4_register_sysfs(sb);
	if (err)
		goto failed_mount7;
	err = ext4_init_orphan_info(sb);
	if (err)
		goto failed_mount8;

	/* 14. 清理孤儿 inode,标记恢复完成 */
	ext4_orphan_cleanup(sb, es);
	ext4_superblock_csum_set(sb);
	if (needs_recovery) {
		ext4_msg(sb, KERN_INFO, "recovery complete");
		err = ext4_mark_recovery_complete(sb, es);
		if (err)
			goto failed_mount10;
	}
	return 0;
	...
}

这段代码覆盖了 ext4 挂载的所有关键阶段:读取超级块、校验元数据、解析选项、初始化块组、加载日志、获取根 inode、建立 s_root、初始化分配器、注册 sysfs、清理孤儿 inode、标记恢复完成。

特别注意 d_make_root(root) 这一行 :它把根 inode 包装成根 dentry,赋给 sb->s_root。这正是上游 vfs_get_tree() 里 BUG_ON(!fc->root) 所期待的最终结果------通过 get_tree_bdev() 里的 fc->root = dget(s->s_root) 传回上下文。

错误处理:goto 链的设计

__ext4_fill_super() 的尾部是一长串 failed_mount* 标签,这是内核错误处理的经典范式:

c

复制代码
failed_mount10:
	ext4_quotas_off(sb, EXT4_MAXQUOTAS);
failed_mount9: __maybe_unused
	ext4_release_orphan_info(sb);
failed_mount8:
	ext4_unregister_sysfs(sb);
	kobject_put(&sbi->s_kobj);
failed_mount7:
	ext4_unregister_li_request(sb);
failed_mount6:
	ext4_mb_release(sb);
	ext4_flex_groups_free(sbi);
	ext4_percpu_param_destroy(sbi);
failed_mount5:
	ext4_ext_release(sb);
	ext4_release_system_zone(sb);
failed_mount4a:
	dput(sb->s_root);
	sb->s_root = NULL;
failed_mount4:
	ext4_msg(sb, KERN_ERR, "mount failed");
	if (EXT4_SB(sb)->rsv_conversion_wq)
		destroy_workqueue(EXT4_SB(sb)->rsv_conversion_wq);
failed_mount_wq:
	ext4_xattr_destroy_cache(sbi->s_ea_inode_cache);
	sbi->s_ea_inode_cache = NULL;
	ext4_xattr_destroy_cache(sbi->s_ea_block_cache);
	sbi->s_ea_block_cache = NULL;
	if (sbi->s_journal) {
		flush_work(&sbi->s_sb_upd_work);
		jbd2_journal_destroy(sbi->s_journal);
		sbi->s_journal = NULL;
	}
	...
out_fail:
	invalidate_bdev(sb->s_bdev);
	sb->s_fs_info = NULL;
	return err;
}

这种"倒序释放"的设计非常精妙:每个错误分支跳转到对应的标签,从该标签开始依次执行所有后续清理 ,直到函数返回。这样既避免了代码重复,又保证了资源按获取的逆序释放。阅读内核代码时,看错误处理标签的排布,就能大致推断出资源获取的顺序和模块间的依赖关系 ------例如 failed_mount6 在 failed_mount5 之前,说明 ext4_mb_init()(对应 failed_mount6)在 ext4_ext_release()(对应 failed_mount5)之前执行。

八、回顾完整调用链

把上面所有环节串起来,一次挂载的完整路径是:

text

复制代码
用户空间 mount("/dev/sda1", "/mnt", "ext4", 0, "data=ordered")
   │
   ▼
SYSCALL_DEFINE5(mount, ...)
   │
   ▼
path_mount() → do_new_mount()
   ├─ get_fs_type("ext4")                     // 查找 file_system_type,必要时加载模块
   ├─ fs_context_for_mount(type, sb_flags)
   │    └─ alloc_fs_context(..., FS_CONTEXT_FOR_MOUNT)
   │         ├─ kzalloc(struct fs_context)
   │         ├─ get_filesystem / get_current_cred / get_net / get_user_ns
   │         └─ ext4_init_fs_context()
   │              ├─ kzalloc(struct ext4_fs_context)
   │              └─ fc->ops = &ext4_context_ops
   ├─ vfs_parse_fs_string(fc, "source", ...)  // 解析 source
   ├─ parse_monolithic_mount_data(fc, data)   // 解析 data 字符串
   ├─ vfs_get_tree(fc)
   │    └─ fc->ops->get_tree(fc)              // 回调到 ext4
   │         └─ ext4_get_tree(fc)
   │              └─ get_tree_bdev(fc, ext4_fill_super)
   │                   ├─ lookup_bdev(fc->source, &dev)
   │                   ├─ sget_dev(fc, dev)
   │                   │    └─ sget_fc(fc, super_s_dev_test, super_s_dev_set)
   │                   │         ├─ 遍历 fs_supers 查找现有超级块
   │                   │         ├─ 若无,alloc_super() 分配
   │                   │         ├─ list_add_tail / hlist_add_head 发布
   │                   │         └─ grab_super 或返回新建的 s
   │                   ├─ setup_bdev_super(s, flags, fc)
   │                   ├─ fill_super(s, fc)   // 即 ext4_fill_super
   │                   │    ├─ ext4_alloc_sbi(sb)
   │                   │    └─ __ext4_fill_super(fc, sb)
   │                   │         ├─ ext4_load_super
   │                   │         ├─ ext4_init_metadata_csum
   │                   │         ├─ ext4_apply_options
   │                   │         ├─ ext4_load_and_init_journal
   │                   │         ├─ ext4_iget(EXT4_ROOT_INO)
   │                   │         ├─ d_make_root(root) → sb->s_root
   │                   │         ├─ ext4_mb_init
   │                   │         ├─ ext4_register_sysfs
   │                   │         └─ ext4_orphan_cleanup
   │                   └─ fc->root = dget(s->s_root)   // 交回上下文
   ├─ super_wake(sb, SB_BORN)                 // 标记超级块就绪
   ├─ security_sb_set_mnt_opts(sb, ...)       // LSM 检查
   └─ do_new_mount_fc(fc, path, mnt_flags)    // 把挂载点加入命名空间

九、设计哲学的思考

读完整条链,我们可以总结出内核挂载框架的几个设计哲学:

第一,抽象分层清晰。 do_new_mount() 只负责"通用流程",fs_context 负责"配置传递",file_system_type 负责"类型描述",super_block 负责"运行时状态"。每一层各司其职,不越界。

第二,回调无处不在。 init_fs_context、parse_param、get_tree、reconfigure、free、test、set------这些回调把"通用框架"与"文件系统特定逻辑"解耦,使 ext4、xfs、btrfs 能共享大部分挂载代码。

第三,错误处理即文档。 goto 标签的排布顺序反映了资源获取顺序;BUG() 与 WARN() 的位置反映了接口契约;注释里的"why"比"what"更被重视。

第四,并发安全优先。 sget_fc() 的"释放锁-分配-重试"模式、sb_lock 短临界区、try_module_get 引用计数,都体现了内核对并发场景的严谨态度。

第五,兼容性渐进。 fc->oldapi 标记、legacy_init_fs_context() 回退、parse_monolithic_mount_data() 兼容传统字符串------新框架在引入的同时,为旧文件系统保留了完整的生存空间。

理解这条调用链,不仅能帮助我们分析挂载失败、死锁、资源泄漏等实际问题的根因,更能让我们体会内核工程师如何在"性能、正确性、可维护性"三者之间做出精妙的权衡。下一次你在代码里加一行调试打印,不妨先想一想:这行代码处于调用链的哪一环?它持有的锁有哪些?它依赖的子系统是否已经就绪?------这些问题的答案,往往就藏在本文剖析的这些函数里。

#源码

cpp 复制代码
/*
 * create a new mount for userspace and request it to be added into the
 * namespace's tree
 */
static int do_new_mount(struct path *path, const char *fstype, int sb_flags,
			int mnt_flags, const char *name, void *data)
{
	struct file_system_type *type;
	struct fs_context *fc;
	const char *subtype = NULL;
	int err = 0;

	if (!fstype)
		return -EINVAL;

	type = get_fs_type(fstype);
	if (!type)
		return -ENODEV;

	if (type->fs_flags & FS_HAS_SUBTYPE) {
		subtype = strchr(fstype, '.');
		if (subtype) {
			subtype++;
			if (!*subtype) {
				put_filesystem(type);
				return -EINVAL;
			}
		}
	}

	fc = fs_context_for_mount(type, sb_flags);
	put_filesystem(type);
	if (IS_ERR(fc))
		return PTR_ERR(fc);

	/*
	 * Indicate to the filesystem that the mount request is coming
	 * from the legacy mount system call.
	 */
	fc->oldapi = true;

	if (subtype)
		err = vfs_parse_fs_string(fc, "subtype",
					  subtype, strlen(subtype));
	if (!err && name)
		err = vfs_parse_fs_string(fc, "source", name, strlen(name));
	if (!err)
		err = parse_monolithic_mount_data(fc, data);
	if (!err && !mount_capable(fc))
		err = -EPERM;
	if (!err)
		err = vfs_get_tree(fc);
	if (!err)
		err = do_new_mount_fc(fc, path, mnt_flags);

	put_fs_context(fc);
	return err;
}

struct fs_context *fs_context_for_mount(struct file_system_type *fs_type,
					unsigned int sb_flags)
{
	return alloc_fs_context(fs_type, NULL, sb_flags, 0,
					FS_CONTEXT_FOR_MOUNT);
}
EXPORT_SYMBOL(fs_context_for_mount);


/**
 * alloc_fs_context - Create a filesystem context.
 * @fs_type: The filesystem type.
 * @reference: The dentry from which this one derives (or NULL)
 * @sb_flags: Filesystem/superblock flags (SB_*)
 * @sb_flags_mask: Applicable members of @sb_flags
 * @purpose: The purpose that this configuration shall be used for.
 *
 * Open a filesystem and create a mount context.  The mount context is
 * initialised with the supplied flags and, if a submount/automount from
 * another superblock (referred to by @reference) is supplied, may have
 * parameters such as namespaces copied across from that superblock.
 */
static struct fs_context *alloc_fs_context(struct file_system_type *fs_type,
				      struct dentry *reference,
				      unsigned int sb_flags,
				      unsigned int sb_flags_mask,
				      enum fs_context_purpose purpose)
{
	int (*init_fs_context)(struct fs_context *);
	struct fs_context *fc;
	int ret = -ENOMEM;

	fc = kzalloc(sizeof(struct fs_context), GFP_KERNEL_ACCOUNT);
	if (!fc)
		return ERR_PTR(-ENOMEM);

	fc->purpose	= purpose;
	fc->sb_flags	= sb_flags;
	fc->sb_flags_mask = sb_flags_mask;
	fc->fs_type	= get_filesystem(fs_type);
	fc->cred	= get_current_cred();
	fc->net_ns	= get_net(current->nsproxy->net_ns);
	fc->log.prefix	= fs_type->name;

	mutex_init(&fc->uapi_mutex);

	switch (purpose) {
	case FS_CONTEXT_FOR_MOUNT:
		fc->user_ns = get_user_ns(fc->cred->user_ns);
		break;
	case FS_CONTEXT_FOR_SUBMOUNT:
		fc->user_ns = get_user_ns(reference->d_sb->s_user_ns);
		break;
	case FS_CONTEXT_FOR_RECONFIGURE:
		atomic_inc(&reference->d_sb->s_active);
		fc->user_ns = get_user_ns(reference->d_sb->s_user_ns);
		fc->root = dget(reference);
		break;
	}

	/* TODO: Make all filesystems support this unconditionally */
	init_fs_context = fc->fs_type->init_fs_context;
	if (!init_fs_context)
		init_fs_context = legacy_init_fs_context;

	ret = init_fs_context(fc);
	if (ret < 0)
		goto err_fc;
	fc->need_free = true;
	return fc;

err_fc:
	put_fs_context(fc);
	return ERR_PTR(ret);
}

struct file_system_type *get_fs_type(const char *name)
{
	struct file_system_type *fs;
	const char *dot = strchr(name, '.');
	int len = dot ? dot - name : strlen(name);

	fs = __get_fs_type(name, len);
	if (!fs && (request_module("fs-%.*s", len, name) == 0)) {
		fs = __get_fs_type(name, len);
		if (!fs)
			pr_warn_once("request_module fs-%.*s succeeded, but still no fs?\n",
				     len, name);
	}

	if (dot && fs && !(fs->fs_flags & FS_HAS_SUBTYPE)) {
		put_filesystem(fs);
		fs = NULL;
	}
	return fs;
}

EXPORT_SYMBOL(get_fs_type);

static struct file_system_type *__get_fs_type(const char *name, int len)
{
	struct file_system_type *fs;

	read_lock(&file_systems_lock);
	fs = *(find_filesystem(name, len));
	if (fs && !try_module_get(fs->owner))
		fs = NULL;
	read_unlock(&file_systems_lock);
	return fs;
}

static struct file_system_type **find_filesystem(const char *name, unsigned len)
{
	struct file_system_type **p;
	for (p = &file_systems; *p; p = &(*p)->next)
		if (strncmp((*p)->name, name, len) == 0 &&
		    !(*p)->name[len])
			break;
	return p;
}


/**
 * vfs_get_tree - Get the mountable root
 * @fc: The superblock configuration context.
 *
 * The filesystem is invoked to get or create a superblock which can then later
 * be used for mounting.  The filesystem places a pointer to the root to be
 * used for mounting in @fc->root.
 */
int vfs_get_tree(struct fs_context *fc)
{
	struct super_block *sb;
	int error;

	if (fc->root)
		return -EBUSY;

	/* Get the mountable root in fc->root, with a ref on the root and a ref
	 * on the superblock.
	 */
	error = fc->ops->get_tree(fc);
	if (error < 0)
		return error;

	if (!fc->root) {
		pr_err("Filesystem %s get_tree() didn't set fc->root\n",
		       fc->fs_type->name);
		/* We don't know what the locking state of the superblock is -
		 * if there is a superblock.
		 */
		BUG();
	}

	sb = fc->root->d_sb;
	WARN_ON(!sb->s_bdi);

	/*
	 * super_wake() contains a memory barrier which also care of
	 * ordering for super_cache_count(). We place it before setting
	 * SB_BORN as the data dependency between the two functions is
	 * the superblock structure contents that we just set up, not
	 * the SB_BORN flag.
	 */
	super_wake(sb, SB_BORN);

	error = security_sb_set_mnt_opts(sb, fc->security, 0, NULL);
	if (unlikely(error)) {
		fc_drop_locked(fc);
		return error;
	}

	/*
	 * filesystems should never set s_maxbytes larger than MAX_LFS_FILESIZE
	 * but s_maxbytes was an unsigned long long for many releases. Throw
	 * this warning for a little while to try and catch filesystems that
	 * violate this rule.
	 */
	WARN((sb->s_maxbytes < 0), "%s set sb->s_maxbytes to "
		"negative value (%lld)\n", fc->fs_type->name, sb->s_maxbytes);

	return 0;
}
EXPORT_SYMBOL(vfs_get_tree);

static int ext4_get_tree(struct fs_context *fc)
{
	return get_tree_bdev(fc, ext4_fill_super);
}


static int __ext4_fill_super(struct fs_context *fc, struct super_block *sb)
{
	struct ext4_super_block *es = NULL;
	struct ext4_sb_info *sbi = EXT4_SB(sb);
	ext4_fsblk_t logical_sb_block;
	struct inode *root;
	int needs_recovery;
	int err;
	ext4_group_t first_not_zeroed;
	struct ext4_fs_context *ctx = fc->fs_private;
	int silent = fc->sb_flags & SB_SILENT;

	/* Set defaults for the variables that will be set during parsing */
	if (!(ctx->spec & EXT4_SPEC_JOURNAL_IOPRIO))
		ctx->journal_ioprio = DEFAULT_JOURNAL_IOPRIO;

	sbi->s_inode_readahead_blks = EXT4_DEF_INODE_READAHEAD_BLKS;
	sbi->s_sectors_written_start =
		part_stat_read(sb->s_bdev, sectors[STAT_WRITE]);

	err = ext4_load_super(sb, &logical_sb_block, silent);
	if (err)
		goto out_fail;

	es = sbi->s_es;
	sbi->s_kbytes_written = le64_to_cpu(es->s_kbytes_written);

	err = ext4_init_metadata_csum(sb, es);
	if (err)
		goto failed_mount;

	ext4_set_def_opts(sb, es);

	sbi->s_resuid = make_kuid(&init_user_ns, le16_to_cpu(es->s_def_resuid));
	sbi->s_resgid = make_kgid(&init_user_ns, le16_to_cpu(es->s_def_resgid));
	sbi->s_commit_interval = JBD2_DEFAULT_MAX_COMMIT_AGE * HZ;
	sbi->s_min_batch_time = EXT4_DEF_MIN_BATCH_TIME;
	sbi->s_max_batch_time = EXT4_DEF_MAX_BATCH_TIME;

	/*
	 * set default s_li_wait_mult for lazyinit, for the case there is
	 * no mount option specified.
	 */
	sbi->s_li_wait_mult = EXT4_DEF_LI_WAIT_MULT;

	err = ext4_inode_info_init(sb, es);
	if (err)
		goto failed_mount;

	err = parse_apply_sb_mount_options(sb, ctx);
	if (err < 0)
		goto failed_mount;

	sbi->s_def_mount_opt = sbi->s_mount_opt;
	sbi->s_def_mount_opt2 = sbi->s_mount_opt2;

	err = ext4_check_opt_consistency(fc, sb);
	if (err < 0)
		goto failed_mount;

	ext4_apply_options(fc, sb);

	err = ext4_encoding_init(sb, es);
	if (err)
		goto failed_mount;

	err = ext4_check_journal_data_mode(sb);
	if (err)
		goto failed_mount;

	sb->s_flags = (sb->s_flags & ~SB_POSIXACL) |
		(test_opt(sb, POSIX_ACL) ? SB_POSIXACL : 0);

	/* i_version is always enabled now */
	sb->s_flags |= SB_I_VERSION;

	err = ext4_check_feature_compatibility(sb, es, silent);
	if (err)
		goto failed_mount;

	err = ext4_block_group_meta_init(sb, silent);
	if (err)
		goto failed_mount;

	ext4_hash_info_init(sb);

	err = ext4_handle_clustersize(sb);
	if (err)
		goto failed_mount;

	err = ext4_check_geometry(sb, es);
	if (err)
		goto failed_mount;

	timer_setup(&sbi->s_err_report, print_daily_error_info, 0);
	spin_lock_init(&sbi->s_error_lock);
	INIT_WORK(&sbi->s_sb_upd_work, update_super_work);

	err = ext4_group_desc_init(sb, es, logical_sb_block, &first_not_zeroed);
	if (err)
		goto failed_mount3;

	err = ext4_es_register_shrinker(sbi);
	if (err)
		goto failed_mount3;

	sbi->s_stripe = ext4_get_stripe_size(sbi);
	/*
	 * It's hard to get stripe aligned blocks if stripe is not aligned with
	 * cluster, just disable stripe and alert user to simpfy code and avoid
	 * stripe aligned allocation which will rarely successes.
	 */
	if (sbi->s_stripe > 0 && sbi->s_cluster_ratio > 1 &&
	    sbi->s_stripe % sbi->s_cluster_ratio != 0) {
		ext4_msg(sb, KERN_WARNING,
			 "stripe (%lu) is not aligned with cluster size (%u), "
			 "stripe is disabled",
			 sbi->s_stripe, sbi->s_cluster_ratio);
		sbi->s_stripe = 0;
	}
	sbi->s_extent_max_zeroout_kb = 32;

	/*
	 * set up enough so that it can read an inode
	 */
	sb->s_op = &ext4_sops;
	sb->s_export_op = &ext4_export_ops;
	sb->s_xattr = ext4_xattr_handlers;
#ifdef CONFIG_FS_ENCRYPTION
	sb->s_cop = &ext4_cryptops;
#endif
#ifdef CONFIG_FS_VERITY
	sb->s_vop = &ext4_verityops;
#endif
#ifdef CONFIG_QUOTA
	sb->dq_op = &ext4_quota_operations;
	if (ext4_has_feature_quota(sb))
		sb->s_qcop = &dquot_quotactl_sysfile_ops;
	else
		sb->s_qcop = &ext4_qctl_operations;
	sb->s_quota_types = QTYPE_MASK_USR | QTYPE_MASK_GRP | QTYPE_MASK_PRJ;
#endif
	memcpy(&sb->s_uuid, es->s_uuid, sizeof(es->s_uuid));

	INIT_LIST_HEAD(&sbi->s_orphan); /* unlinked but open files */
	mutex_init(&sbi->s_orphan_lock);

	ext4_fast_commit_init(sb);

	sb->s_root = NULL;

	needs_recovery = (es->s_last_orphan != 0 ||
			  ext4_has_feature_orphan_present(sb) ||
			  ext4_has_feature_journal_needs_recovery(sb));

	if (ext4_has_feature_mmp(sb) && !sb_rdonly(sb)) {
		err = ext4_multi_mount_protect(sb, le64_to_cpu(es->s_mmp_block));
		if (err)
			goto failed_mount3a;
	}

	err = -EINVAL;
	/*
	 * The first inode we look at is the journal inode.  Don't try
	 * root first: it may be modified in the journal!
	 */
	if (!test_opt(sb, NOLOAD) && ext4_has_feature_journal(sb)) {
		err = ext4_load_and_init_journal(sb, es, ctx);
		if (err)
			goto failed_mount3a;
	} else if (test_opt(sb, NOLOAD) && !sb_rdonly(sb) &&
		   ext4_has_feature_journal_needs_recovery(sb)) {
		ext4_msg(sb, KERN_ERR, "required journal recovery "
		       "suppressed and not mounted read-only");
		goto failed_mount3a;
	} else {
		/* Nojournal mode, all journal mount options are illegal */
		if (test_opt(sb, JOURNAL_ASYNC_COMMIT)) {
			ext4_msg(sb, KERN_ERR, "can't mount with "
				 "journal_async_commit, fs mounted w/o journal");
			goto failed_mount3a;
		}

		if (test_opt2(sb, EXPLICIT_JOURNAL_CHECKSUM)) {
			ext4_msg(sb, KERN_ERR, "can't mount with "
				 "journal_checksum, fs mounted w/o journal");
			goto failed_mount3a;
		}
		if (sbi->s_commit_interval != JBD2_DEFAULT_MAX_COMMIT_AGE*HZ) {
			ext4_msg(sb, KERN_ERR, "can't mount with "
				 "commit=%lu, fs mounted w/o journal",
				 sbi->s_commit_interval / HZ);
			goto failed_mount3a;
		}
		if (EXT4_MOUNT_DATA_FLAGS &
		    (sbi->s_mount_opt ^ sbi->s_def_mount_opt)) {
			ext4_msg(sb, KERN_ERR, "can't mount with "
				 "data=, fs mounted w/o journal");
			goto failed_mount3a;
		}
		sbi->s_def_mount_opt &= ~EXT4_MOUNT_JOURNAL_CHECKSUM;
		clear_opt(sb, JOURNAL_CHECKSUM);
		clear_opt(sb, DATA_FLAGS);
		clear_opt2(sb, JOURNAL_FAST_COMMIT);
		sbi->s_journal = NULL;
		needs_recovery = 0;
	}

	if (!test_opt(sb, NO_MBCACHE)) {
		sbi->s_ea_block_cache = ext4_xattr_create_cache();
		if (!sbi->s_ea_block_cache) {
			ext4_msg(sb, KERN_ERR,
				 "Failed to create ea_block_cache");
			err = -EINVAL;
			goto failed_mount_wq;
		}

		if (ext4_has_feature_ea_inode(sb)) {
			sbi->s_ea_inode_cache = ext4_xattr_create_cache();
			if (!sbi->s_ea_inode_cache) {
				ext4_msg(sb, KERN_ERR,
					 "Failed to create ea_inode_cache");
				err = -EINVAL;
				goto failed_mount_wq;
			}
		}
	}

	/*
	 * Get the # of file system overhead blocks from the
	 * superblock if present.
	 */
	sbi->s_overhead = le32_to_cpu(es->s_overhead_clusters);
	/* ignore the precalculated value if it is ridiculous */
	if (sbi->s_overhead > ext4_blocks_count(es))
		sbi->s_overhead = 0;
	/*
	 * If the bigalloc feature is not enabled recalculating the
	 * overhead doesn't take long, so we might as well just redo
	 * it to make sure we are using the correct value.
	 */
	if (!ext4_has_feature_bigalloc(sb))
		sbi->s_overhead = 0;
	if (sbi->s_overhead == 0) {
		err = ext4_calculate_overhead(sb);
		if (err)
			goto failed_mount_wq;
	}

	/*
	 * The maximum number of concurrent works can be high and
	 * concurrency isn't really necessary.  Limit it to 1.
	 */
	EXT4_SB(sb)->rsv_conversion_wq =
		alloc_workqueue("ext4-rsv-conversion", WQ_MEM_RECLAIM | WQ_UNBOUND, 1);
	if (!EXT4_SB(sb)->rsv_conversion_wq) {
		printk(KERN_ERR "EXT4-fs: failed to create workqueue\n");
		err = -ENOMEM;
		goto failed_mount4;
	}

	/*
	 * The jbd2_journal_load will have done any necessary log recovery,
	 * so we can safely mount the rest of the filesystem now.
	 */

	root = ext4_iget(sb, EXT4_ROOT_INO, EXT4_IGET_SPECIAL);
	if (IS_ERR(root)) {
		ext4_msg(sb, KERN_ERR, "get root inode failed");
		err = PTR_ERR(root);
		root = NULL;
		goto failed_mount4;
	}
	if (!S_ISDIR(root->i_mode) || !root->i_blocks || !root->i_size) {
		ext4_msg(sb, KERN_ERR, "corrupt root inode, run e2fsck");
		iput(root);
		err = -EFSCORRUPTED;
		goto failed_mount4;
	}

	sb->s_root = d_make_root(root);
	if (!sb->s_root) {
		ext4_msg(sb, KERN_ERR, "get root dentry failed");
		err = -ENOMEM;
		goto failed_mount4;
	}

	err = ext4_setup_super(sb, es, sb_rdonly(sb));
	if (err == -EROFS) {
		sb->s_flags |= SB_RDONLY;
	} else if (err)
		goto failed_mount4a;

	ext4_set_resv_clusters(sb);

	if (test_opt(sb, BLOCK_VALIDITY)) {
		err = ext4_setup_system_zone(sb);
		if (err) {
			ext4_msg(sb, KERN_ERR, "failed to initialize system "
				 "zone (%d)", err);
			goto failed_mount4a;
		}
	}
	ext4_fc_replay_cleanup(sb);

	ext4_ext_init(sb);

	/*
	 * Enable optimize_scan if number of groups is > threshold. This can be
	 * turned off by passing "mb_optimize_scan=0". This can also be
	 * turned on forcefully by passing "mb_optimize_scan=1".
	 */
	if (!(ctx->spec & EXT4_SPEC_mb_optimize_scan)) {
		if (sbi->s_groups_count >= MB_DEFAULT_LINEAR_SCAN_THRESHOLD)
			set_opt2(sb, MB_OPTIMIZE_SCAN);
		else
			clear_opt2(sb, MB_OPTIMIZE_SCAN);
	}

	err = ext4_mb_init(sb);
	if (err) {
		ext4_msg(sb, KERN_ERR, "failed to initialize mballoc (%d)",
			 err);
		goto failed_mount5;
	}

	/*
	 * We can only set up the journal commit callback once
	 * mballoc is initialized
	 */
	if (sbi->s_journal)
		sbi->s_journal->j_commit_callback =
			ext4_journal_commit_callback;

	err = ext4_percpu_param_init(sbi);
	if (err)
		goto failed_mount6;

	if (ext4_has_feature_flex_bg(sb))
		if (!ext4_fill_flex_info(sb)) {
			ext4_msg(sb, KERN_ERR,
			       "unable to initialize "
			       "flex_bg meta info!");
			err = -ENOMEM;
			goto failed_mount6;
		}

	err = ext4_register_li_request(sb, first_not_zeroed);
	if (err)
		goto failed_mount6;

	err = ext4_register_sysfs(sb);
	if (err)
		goto failed_mount7;

	err = ext4_init_orphan_info(sb);
	if (err)
		goto failed_mount8;
#ifdef CONFIG_QUOTA
	/* Enable quota usage during mount. */
	if (ext4_has_feature_quota(sb) && !sb_rdonly(sb)) {
		err = ext4_enable_quotas(sb);
		if (err)
			goto failed_mount9;
	}
#endif  /* CONFIG_QUOTA */

	/*
	 * Save the original bdev mapping's wb_err value which could be
	 * used to detect the metadata async write error.
	 */
	spin_lock_init(&sbi->s_bdev_wb_lock);
	errseq_check_and_advance(&sb->s_bdev->bd_inode->i_mapping->wb_err,
				 &sbi->s_bdev_wb_err);
	EXT4_SB(sb)->s_mount_state |= EXT4_ORPHAN_FS;
	ext4_orphan_cleanup(sb, es);
	EXT4_SB(sb)->s_mount_state &= ~EXT4_ORPHAN_FS;
	/*
	 * Update the checksum after updating free space/inode counters and
	 * ext4_orphan_cleanup. Otherwise the superblock can have an incorrect
	 * checksum in the buffer cache until it is written out and
	 * e2fsprogs programs trying to open a file system immediately
	 * after it is mounted can fail.
	 */
	ext4_superblock_csum_set(sb);
	if (needs_recovery) {
		ext4_msg(sb, KERN_INFO, "recovery complete");
		err = ext4_mark_recovery_complete(sb, es);
		if (err)
			goto failed_mount10;
	}

	if (test_opt(sb, DISCARD) && !bdev_max_discard_sectors(sb->s_bdev))
		ext4_msg(sb, KERN_WARNING,
			 "mounting with \"discard\" option, but the device does not support discard");

	if (es->s_error_count)
		mod_timer(&sbi->s_err_report, jiffies + 300*HZ); /* 5 minutes */

	/* Enable message ratelimiting. Default is 10 messages per 5 secs. */
	ratelimit_state_init(&sbi->s_err_ratelimit_state, 5 * HZ, 10);
	ratelimit_state_init(&sbi->s_warning_ratelimit_state, 5 * HZ, 10);
	ratelimit_state_init(&sbi->s_msg_ratelimit_state, 5 * HZ, 10);
	atomic_set(&sbi->s_warning_count, 0);
	atomic_set(&sbi->s_msg_count, 0);

	return 0;

failed_mount10:
	ext4_quotas_off(sb, EXT4_MAXQUOTAS);
failed_mount9: __maybe_unused
	ext4_release_orphan_info(sb);
failed_mount8:
	ext4_unregister_sysfs(sb);
	kobject_put(&sbi->s_kobj);
failed_mount7:
	ext4_unregister_li_request(sb);
failed_mount6:
	ext4_mb_release(sb);
	ext4_flex_groups_free(sbi);
	ext4_percpu_param_destroy(sbi);
failed_mount5:
	ext4_ext_release(sb);
	ext4_release_system_zone(sb);
failed_mount4a:
	dput(sb->s_root);
	sb->s_root = NULL;
failed_mount4:
	ext4_msg(sb, KERN_ERR, "mount failed");
	if (EXT4_SB(sb)->rsv_conversion_wq)
		destroy_workqueue(EXT4_SB(sb)->rsv_conversion_wq);
failed_mount_wq:
	ext4_xattr_destroy_cache(sbi->s_ea_inode_cache);
	sbi->s_ea_inode_cache = NULL;

	ext4_xattr_destroy_cache(sbi->s_ea_block_cache);
	sbi->s_ea_block_cache = NULL;

	if (sbi->s_journal) {
		/* flush s_sb_upd_work before journal destroy. */
		flush_work(&sbi->s_sb_upd_work);
		jbd2_journal_destroy(sbi->s_journal);
		sbi->s_journal = NULL;
	}
failed_mount3a:
	ext4_es_unregister_shrinker(sbi);
failed_mount3:
	/* flush s_sb_upd_work before sbi destroy */
	flush_work(&sbi->s_sb_upd_work);
	del_timer_sync(&sbi->s_err_report);
	ext4_stop_mmpd(sbi);
	ext4_group_desc_free(sbi);
failed_mount:
	if (sbi->s_chksum_driver)
		crypto_free_shash(sbi->s_chksum_driver);

#if IS_ENABLED(CONFIG_UNICODE)
	utf8_unload(sb->s_encoding);
#endif

#ifdef CONFIG_QUOTA
	for (unsigned int i = 0; i < EXT4_MAXQUOTAS; i++)
		kfree(get_qf_name(sb, sbi, i));
#endif
	fscrypt_free_dummy_policy(&sbi->s_dummy_enc_policy);
	brelse(sbi->s_sbh);
	if (sbi->s_journal_bdev_handle) {
		invalidate_bdev(sbi->s_journal_bdev_handle->bdev);
		bdev_release(sbi->s_journal_bdev_handle);
	}
out_fail:
	invalidate_bdev(sb->s_bdev);
	sb->s_fs_info = NULL;
	return err;
}

static int ext4_fill_super(struct super_block *sb, struct fs_context *fc)
{
	struct ext4_fs_context *ctx = fc->fs_private;
	struct ext4_sb_info *sbi;
	const char *descr;
	int ret;

	sbi = ext4_alloc_sbi(sb);
	if (!sbi)
		return -ENOMEM;

	fc->s_fs_info = sbi;

	/* Cleanup superblock name */
	strreplace(sb->s_id, '/', '!');

	sbi->s_sb_block = 1;	/* Default super block location */
	if (ctx->spec & EXT4_SPEC_s_sb_block)
		sbi->s_sb_block = ctx->s_sb_block;

	ret = __ext4_fill_super(fc, sb);
	//yym-gaizao
	pr_info("YYM-DBG: [ext4_fill_super] Attempting to fill ext4 superblock for device %s \n", sb->s_id);
	if (ret < 0)
		goto free_sbi;

	if (sbi->s_journal) {
		if (test_opt(sb, DATA_FLAGS) == EXT4_MOUNT_JOURNAL_DATA)
			descr = " journalled data mode";
		else if (test_opt(sb, DATA_FLAGS) == EXT4_MOUNT_ORDERED_DATA)
			descr = " ordered data mode";
		else
			descr = " writeback data mode";
	} else
		descr = "out journal";

	if (___ratelimit(&ext4_mount_msg_ratelimit, "EXT4-fs mount"))
		ext4_msg(sb, KERN_INFO, "mounted filesystem %pU %s with%s. "
			 "Quota mode: %s.", &sb->s_uuid,
			 sb_rdonly(sb) ? "ro" : "r/w", descr,
			 ext4_quota_mode(sb));

	/* Update the s_overhead_clusters if necessary */
	ext4_update_overhead(sb, false);
	return 0;

free_sbi:
	ext4_free_sbi(sbi);
	fc->s_fs_info = NULL;
	return ret;
}

static const struct fs_context_operations ext4_context_ops = {
	.parse_param	= ext4_parse_param,
	.get_tree	= ext4_get_tree,
	.reconfigure	= ext4_reconfigure,
	.free		= ext4_fc_free,
};

int ext4_init_fs_context(struct fs_context *fc)
{
	struct ext4_fs_context *ctx;

	ctx = kzalloc(sizeof(struct ext4_fs_context), GFP_KERNEL);
	if (!ctx)
		return -ENOMEM;

	fc->fs_private = ctx;
	fc->ops = &ext4_context_ops;

	return 0;
}

/**
 * get_tree_bdev - Get a superblock based on a single block device
 * @fc: The filesystem context holding the parameters
 * @fill_super: Helper to initialise a new superblock
 */
int get_tree_bdev(struct fs_context *fc,
		int (*fill_super)(struct super_block *,
				  struct fs_context *))
{
	struct super_block *s;
	int error = 0;
	dev_t dev;

	if (!fc->source)
		return invalf(fc, "No source specified");

	error = lookup_bdev(fc->source, &dev);
	if (error) {
		errorf(fc, "%s: Can't lookup blockdev", fc->source);
		return error;
	}

	fc->sb_flags |= SB_NOSEC;
	s = sget_dev(fc, dev);
	if (IS_ERR(s))
		return PTR_ERR(s);

	if (s->s_root) {
		/* Don't summarily change the RO/RW state. */
		if ((fc->sb_flags ^ s->s_flags) & SB_RDONLY) {
			warnf(fc, "%pg: Can't mount, would change RO state", s->s_bdev);
			deactivate_locked_super(s);
			return -EBUSY;
		}
	} else {
		error = setup_bdev_super(s, fc->sb_flags, fc);
		if (!error)
			error = fill_super(s, fc);
		if (error) {
			deactivate_locked_super(s);
			return error;
		}
		s->s_flags |= SB_ACTIVE;
	}

	BUG_ON(fc->root);
	fc->root = dget(s->s_root);
	return 0;
}
EXPORT_SYMBOL(get_tree_bdev);

/**
 * sget_dev - Find or create a superblock by device number
 * @fc: Filesystem context.
 * @dev: device number
 *
 * Find or create a superblock using the provided device number that
 * will be stored in fc->sget_key.
 *
 * If an extant superblock is matched, then that will be returned with
 * an elevated reference count that the caller must transfer or discard.
 *
 * If no match is made, a new superblock will be allocated and basic
 * initialisation will be performed (s_type, s_fs_info, s_id, s_dev will
 * be set). The superblock will be published and it will be returned in
 * a partially constructed state with SB_BORN and SB_ACTIVE as yet
 * unset.
 *
 * Return: an existing or newly created superblock on success, an error
 *         pointer on failure.
 */
struct super_block *sget_dev(struct fs_context *fc, dev_t dev)
{
	fc->sget_key = &dev;
	return sget_fc(fc, super_s_dev_test, super_s_dev_set);
}
EXPORT_SYMBOL(sget_dev);

/**
 * sget_fc - Find or create a superblock
 * @fc:	Filesystem context.
 * @test: Comparison callback
 * @set: Setup callback
 *
 * Create a new superblock or find an existing one.
 *
 * The @test callback is used to find a matching existing superblock.
 * Whether or not the requested parameters in @fc are taken into account
 * is specific to the @test callback that is used. They may even be
 * completely ignored.
 *
 * If an extant superblock is matched, it will be returned unless:
 *
 * (1) the namespace the filesystem context @fc and the extant
 *     superblock's namespace differ
 *
 * (2) the filesystem context @fc has requested that reusing an extant
 *     superblock is not allowed
 *
 * In both cases EBUSY will be returned.
 *
 * If no match is made, a new superblock will be allocated and basic
 * initialisation will be performed (s_type, s_fs_info and s_id will be
 * set and the @set callback will be invoked), the superblock will be
 * published and it will be returned in a partially constructed state
 * with SB_BORN and SB_ACTIVE as yet unset.
 *
 * Return: On success, an extant or newly created superblock is
 *         returned. On failure an error pointer is returned.
 */
struct super_block *sget_fc(struct fs_context *fc,
			    int (*test)(struct super_block *, struct fs_context *),
			    int (*set)(struct super_block *, struct fs_context *))
{
	struct super_block *s = NULL;
	struct super_block *old;
	struct user_namespace *user_ns = fc->global ? &init_user_ns : fc->user_ns;
	int err;

retry:
	spin_lock(&sb_lock);
	if (test) {
		hlist_for_each_entry(old, &fc->fs_type->fs_supers, s_instances) {
			if (test(old, fc))
				goto share_extant_sb;
		}
	}
	if (!s) {
		spin_unlock(&sb_lock);
		s = alloc_super(fc->fs_type, fc->sb_flags, user_ns);
		if (!s)
			return ERR_PTR(-ENOMEM);
		goto retry;
	}

	s->s_fs_info = fc->s_fs_info;
	err = set(s, fc);
	if (err) {
		s->s_fs_info = NULL;
		spin_unlock(&sb_lock);
		destroy_unused_super(s);
		return ERR_PTR(err);
	}
	fc->s_fs_info = NULL;
	s->s_type = fc->fs_type;
	s->s_iflags |= fc->s_iflags;
	strscpy(s->s_id, s->s_type->name, sizeof(s->s_id));
	/*
	 * Make the superblock visible on @super_blocks and @fs_supers.
	 * It's in a nascent state and users should wait on SB_BORN or
	 * SB_DYING to be set.
	 */
	list_add_tail(&s->s_list, &super_blocks);
	hlist_add_head(&s->s_instances, &s->s_type->fs_supers);
	spin_unlock(&sb_lock);
	get_filesystem(s->s_type);
	shrinker_register(s->s_shrink);
	return s;

share_extant_sb:
	if (user_ns != old->s_user_ns || fc->exclusive) {
		spin_unlock(&sb_lock);
		destroy_unused_super(s);
		if (fc->exclusive)
			warnfc(fc, "reusing existing filesystem not allowed");
		else
			warnfc(fc, "reusing existing filesystem in another namespace not allowed");
		return ERR_PTR(-EBUSY);
	}
	if (!grab_super(old))
		goto retry;
	destroy_unused_super(s);
	return old;
}
EXPORT_SYMBOL(sget_fc);
相关推荐
害人终害己1 小时前
Redis 日志:AUTH 认证信息
运维·服务器
a努力。1 小时前
Context-State-Memory三重信息架构揭秘
java·服务器·前端
天蓝蓝的本我1 小时前
linux本地部署Qwen3.8 27B
linux·运维·excel
n112122 小时前
服务器被反复尝试登录之后:fail2ban 加密钥登录的五道加固
运维·服务器·服务器安全·fail2ban·ssh加固
洋不写bug2 小时前
网络编程(二)TCP回显服务器与客户端通信详解
服务器·网络·tcp/ip·tcp·网络通信·javaee·回显服务器
java_logo2 小时前
Docker 部署 DeepSeek Harness:轻松搭建局域网里的 AI Agent 平台
运维·docker·容器·ai agent·deepseek·轩辕镜像·deepseekharness
lisanmengmeng2 小时前
NRPE 添加命令(四)
linux·运维·服务器
想做小南娘,发现自己是女生喵2 小时前
i.MX6ULL嵌入式Linux入门学习文档
linux·运维·学习
FL16238631292 小时前
Siemens‑Automation‑License‑Manager & SIMATIC‑Automation‑Tool 下载安装教程
运维·服务器