挂载(mount)是文件系统与用户空间之间的第一座桥梁,也是内核中最容易出问题、最难调试的路径之一。Linux 内核在近几个版本中对挂载 API 做了大规模重构,引入了以 fs_context 为中心的新挂载框架,把"参数解析"与"超级块获取"从传统 mount() 系统调用的泥潭里剥离出来。本文以 Linux 6.18 源码为蓝本,沿着 do_new_mount() → fs_context_for_mount() → vfs_get_tree() → ext4_get_tree() → get_tree_bdev() → sget_fc() → ext4_fill_super() → __ext4_fill_super() 这条主链,逐层剖析一次挂载请求是如何从用户空间落地到 ext4 超级块的。
一、入口层:do_new_mount 的四个阶段
一切从 do_new_mount() 开始。它是传统 mount() 系统调用进入 VFS 后的核心函数之一,也是新旧挂载 API 的交汇点:
c
static int do_new_mount(struct path *path, const char *fstype, int sb_flags,
int mnt_flags, const char *name, void *data)
{
struct file_system_type *type;
struct fs_context *fc;
const char *subtype = NULL;
int err = 0;
if (!fstype)
return -EINVAL;
type = get_fs_type(fstype); // 【阶段 1】查找文件系统类型
if (!type)
return -ENODEV;
if (type->fs_flags & FS_HAS_SUBTYPE) {
subtype = strchr(fstype, '.'); // 处理 "ext4.dev" 这类子类型
if (subtype) {
subtype++;
if (!*subtype) {
put_filesystem(type);
return -EINVAL;
}
}
}
fc = fs_context_for_mount(type, sb_flags); // 【阶段 2】创建上下文
put_filesystem(type);
if (IS_ERR(fc))
return PTR_ERR(fc);
/*
* Indicate to the filesystem that the mount request is coming
* from the legacy mount system call.
*/
fc->oldapi = true; // 标记来自传统 mount() 调用
if (subtype)
err = vfs_parse_fs_string(fc, "subtype",
subtype, strlen(subtype));
if (!err && name)
err = vfs_parse_fs_string(fc, "source", name, strlen(name));
if (!err)
err = parse_monolithic_mount_data(fc, data); // 【阶段 3】解析参数
if (!err && !mount_capable(fc))
err = -EPERM;
if (!err)
err = vfs_get_tree(fc); // 【阶段 4a】获取超级块
if (!err)
err = do_new_mount_fc(fc, path, mnt_flags); // 【阶段 4b】加入命名空间
put_fs_context(fc);
return err;
}
这段代码的结构非常清晰,可以概括为四个阶段:
-
类型查找 :
get_fs_type()根据用户传入的字符串(如"ext4")找到对应的struct file_system_type。 -
上下文创建 :
fs_context_for_mount()分配并初始化struct fs_context,绑定文件系统的操作集。 -
参数解析 :通过
vfs_parse_fs_string()与parse_monolithic_mount_data()把用户参数填入上下文。 -
超级块获取与挂载 :
vfs_get_tree()触发文件系统去获取或创建超级块,do_new_mount_fc()把挂载点加入当前命名空间。
值得注意的是,fc->oldapi = true 这一行标记了上下文来自旧 API,文件系统可以据此决定是否走兼容路径。这正是新框架"新旧兼容"的设计精髓------fs_context 作为统一的配置载体,既支持新式的"逐参数解析",也支持传统的一整块 data 字符串。
二、类型查找:get_fs_type 与模块自动加载
挂载的第一步是找到"我要挂哪种文件系统"。这个任务由 get_fs_type() 承担:
c
struct file_system_type *get_fs_type(const char *name)
{
struct file_system_type *fs;
const char *dot = strchr(name, '.');
int len = dot ? dot - name : strlen(name);
fs = __get_fs_type(name, len); // 先在已注册链表里找
if (!fs && (request_module("fs-%.*s", len, name) == 0)) {
fs = __get_fs_type(name, len); // 尝试按需加载模块后再找
if (!fs)
pr_warn_once("request_module fs-%.*s succeeded, but still no fs?\n",
len, name);
}
if (dot && fs && !(fs->fs_flags & FS_HAS_SUBTYPE)) {
put_filesystem(fs); // 类型不支持子类型则拒绝
fs = NULL;
}
return fs;
}
这里有一个容易被忽略的细节:request_module() 会在用户态调用 modprobe,尝试加载名为 fs-ext4 之类的模块。如果模块加载成功但仍未注册文件系统,内核会打印一次警告。这个"先查找、后加载、再查找"的双阶段逻辑保证了按需加载与并发安全。
__get_fs_type() 在 file_systems_lock 保护下,通过 find_filesystem() 遍历全局 file_systems 链表,并用 try_module_get() 增加模块引用计数:
c
static struct file_system_type *__get_fs_type(const char *name, int len)
{
struct file_system_type *fs;
read_lock(&file_systems_lock);
fs = *(find_filesystem(name, len));
if (fs && !try_module_get(fs->owner))
fs = NULL;
read_unlock(&file_systems_lock);
return fs;
}
try_module_get() 是关键------它保证了文件系统模块在挂载过程中不会被卸载。对应的 put_filesystem() 在 do_new_mount() 里会在 fs_context_for_mount() 之后调用,意味着文件系统的生命周期由 fs_context 接管。
三、上下文创建:fs_context_for_mount 与 alloc_fs_context
拿到 file_system_type 后,do_new_mount() 调用 fs_context_for_mount(),后者只是 alloc_fs_context() 的薄封装:
c
struct fs_context *fs_context_for_mount(struct file_system_type *fs_type,
unsigned int sb_flags)
{
return alloc_fs_context(fs_type, NULL, sb_flags, 0,
FS_CONTEXT_FOR_MOUNT);
}
alloc_fs_context() 是核心,它承担了上下文初始化的大部分工作:
c
static struct fs_context *alloc_fs_context(struct file_system_type *fs_type,
struct dentry *reference,
unsigned int sb_flags,
unsigned int sb_flags_mask,
enum fs_context_purpose purpose)
{
int (*init_fs_context)(struct fs_context *);
struct fs_context *fc;
int ret = -ENOMEM;
fc = kzalloc(sizeof(struct fs_context), GFP_KERNEL_ACCOUNT);
if (!fc)
return ERR_PTR(-ENOMEM);
fc->purpose = purpose;
fc->sb_flags = sb_flags;
fc->sb_flags_mask = sb_flags_mask;
fc->fs_type = get_filesystem(fs_type);
fc->cred = get_current_cred(); // 保存当前进程的 cred
fc->net_ns = get_net(current->nsproxy->net_ns);
fc->log.prefix = fs_type->name;
mutex_init(&fc->uapi_mutex);
switch (purpose) {
case FS_CONTEXT_FOR_MOUNT:
fc->user_ns = get_user_ns(fc->cred->user_ns); // 使用当前用户命名空间
break;
case FS_CONTEXT_FOR_SUBMOUNT:
fc->user_ns = get_user_ns(reference->d_sb->s_user_ns);
break;
case FS_CONTEXT_FOR_RECONFIGURE:
atomic_inc(&reference->d_sb->s_active);
fc->user_ns = get_user_ns(reference->d_sb->s_user_ns);
fc->root = dget(reference);
break;
}
/* TODO: Make all filesystems support this unconditionally */
init_fs_context = fc->fs_type->init_fs_context;
if (!init_fs_context)
init_fs_context = legacy_init_fs_context; // 旧文件系统走兼容路径
ret = init_fs_context(fc);
if (ret < 0)
goto err_fc;
fc->need_free = true;
return fc;
err_fc:
put_fs_context(fc);
return ERR_PTR(ret);
}
这段代码体现了新挂载框架的三个设计要点:
第一,上下文与用途绑定。 purpose 有三种取值------FOR_MOUNT(新挂载)、FOR_SUBMOUNT(子挂载/自动挂载)、FOR_RECONFIGURE(重新配置)。不同用途决定了 user_ns 的取值来源:新挂载使用当前进程的用户命名空间,子挂载与重配置则继承超级块的用户命名空间。这直接影响后续的权限检查。
第二,init_fs_context 回调是扩展点。 新式文件系统实现了自己的 init_fs_context,可以在此时分配私有数据、绑定操作集。旧文件系统没有这个回调,内核回退到 legacy_init_fs_context(),由后者分配一个兼容上下文。
第三,资源引用在上下文里集中管理。 get_filesystem()、get_current_cred()、get_net()、get_user_ns() 都在这里获取,对应的 put_fs_context() 统一释放,避免了"半路出错忘记释放"的经典问题。
对 ext4 来说,init_fs_context 指向 ext4_init_fs_context():
c
int ext4_init_fs_context(struct fs_context *fc)
{
struct ext4_fs_context *ctx;
ctx = kzalloc(sizeof(struct ext4_fs_context), GFP_KERNEL);
if (!ctx)
return -ENOMEM;
fc->fs_private = ctx; // 私有数据:ext4 参数解析状态
fc->ops = &ext4_context_ops; // 绑定 ext4 的操作集
return 0;
}
ext4_context_ops 定义了 ext4 在挂载过程中的四个关键回调:
c
static const struct fs_context_operations ext4_context_ops = {
.parse_param = ext4_parse_param, // 解析单个挂载参数
.get_tree = ext4_get_tree, // 获取超级块根
.reconfigure = ext4_reconfigure, // 重新配置
.free = ext4_fc_free, // 释放私有数据
};
至此,fs_context 已经完全就绪:它知道"要挂什么类型(fs_type)"、"用什么权限(cred、user_ns)"、"由谁处理(ops)"。
四、获取挂载树:vfs_get_tree 的承上启下
vfs_get_tree() 是 VFS 层获取超级块根的核心函数,也是连接通用框架与文件系统实现的桥梁:
c
int vfs_get_tree(struct fs_context *fc)
{
struct super_block *sb;
int error;
if (fc->root)
return -EBUSY; // 已经获取过,拒绝重复
/* Get the mountable root in fc->root, with a ref on the root and a ref
* on the superblock.
*/
error = fc->ops->get_tree(fc); // 【关键】调用文件系统特定实现
if (error < 0)
return error;
if (!fc->root) {
pr_err("Filesystem %s get_tree() didn't set fc->root\n",
fc->fs_type->name);
BUG(); // 契约:get_tree 必须设置 fc->root
}
sb = fc->root->d_sb;
WARN_ON(!sb->s_bdi);
/*
* super_wake() contains a memory barrier which also care of
* ordering for super_cache_count(). We place it before setting
* SB_BORN as the data dependency between the two functions is
* the superblock structure contents that we just set up, not
* the SB_BORN flag.
*/
super_wake(sb, SB_BORN); // 唤醒超级块,标记 SB_BORN
error = security_sb_set_mnt_opts(sb, fc->security, 0, NULL);
if (unlikely(error)) {
fc_drop_locked(fc);
return error;
}
WARN((sb->s_maxbytes < 0), "%s set sb->s_maxbytes to "
"negative value (%lld)\n", fc->fs_type->name, sb->s_maxbytes);
return 0;
}
这段代码有几个值得关注的细节:
契约式的 BUG() :如果文件系统实现了 get_tree 却没有设置 fc->root,内核直接 BUG()。这不是"防御性编程",而是接口契约的强制执行 ------get_tree 的唯一职责就是设置根 dentry,做不到就是文件系统的错,没有回退余地。
super_wake(sb, SB_BORN) 与内存屏障 :注释里专门解释了为什么要放在 SB_BORN 之前------super_wake() 内含内存屏障,它保护的是 super_cache_count() 的顺序。由于数据依赖是超级块结构体的内容(刚设置好),而不是 SB_BORN 标志本身,所以顺序必须是这样。这种"注释即文档"的风格在内核里很常见,也是阅读内核源码时最值得留意的部分。
security_sb_set_mnt_opts() :LSM(如 SELinux)在此刻介入,设置挂载点的安全选项。如果 LSM 拒绝,fc_drop_locked() 会释放已获取的根与超级块,整个挂载优雅失败。
五、ext4 的实现:ext4_get_tree 与 get_tree_bdev
vfs_get_tree() 通过 fc->ops->get_tree(fc) 回调到 ext4 的 ext4_get_tree():
c
static int ext4_get_tree(struct fs_context *fc)
{
return get_tree_bdev(fc, ext4_fill_super); // 委托给块设备通用辅助
}
ext4 作为块设备文件系统,把大部分逻辑委托给了 get_tree_bdev()。这个函数是所有基于块设备的文件系统的公共辅助:
c
int get_tree_bdev(struct fs_context *fc,
int (*fill_super)(struct super_block *,
struct fs_context *))
{
struct super_block *s;
int error = 0;
dev_t dev;
if (!fc->source)
return invalf(fc, "No source specified");
error = lookup_bdev(fc->source, &dev); // 设备路径 → dev_t
if (error) {
errorf(fc, "%s: Can't lookup blockdev", fc->source);
return error;
}
fc->sb_flags |= SB_NOSEC;
s = sget_dev(fc, dev); // 查找或创建超级块
if (IS_ERR(s))
return PTR_ERR(s);
if (s->s_root) {
/* Don't summarily change the RO/RW state. */
if ((fc->sb_flags ^ s->s_flags) & SB_RDONLY) {
warnf(fc, "%pg: Can't mount, would change RO state", s->s_bdev);
deactivate_locked_super(s);
return -EBUSY;
}
} else {
error = setup_bdev_super(s, fc->sb_flags, fc);
if (!error)
error = fill_super(s, fc); // 【关键】调用 ext4_fill_super
if (error) {
deactivate_locked_super(s);
return error;
}
s->s_flags |= SB_ACTIVE;
}
BUG_ON(fc->root);
fc->root = dget(s->s_root); // 设置可挂载根
return 0;
}
这段代码的逻辑分支非常关键,涉及"超级块是否已存在"两种情况:
情况一:超级块已存在(s->s_root 非空)。 说明该设备已被挂载过。此时不能简单地重新走一遍 fill_super,只需要检查只读状态是否冲突。(fc->sb_flags ^ s->s_flags) & SB_RDONLY 这个异或运算很巧妙------它只关心 SB_RDONLY 这一位是否不同,如果不同(比如原来是 r/o 现在想挂 r/w),则拒绝并返回 -EBUSY。这防止了"先 ro 后 rw 挂同一个设备"造成的状态不一致。
情况二:超级块是新建的。 依次执行 setup_bdev_super()(设置 s_bdev、s_bdi 等基本字段)和 fill_super()(即 ext4_fill_super),成功后设置 SB_ACTIVE。
无论哪种情况,最后都通过 fc->root = dget(s->s_root) 把根 dentry 交给上下文。这个 BUG_ON(fc->root) 是防御性检查------如果之前的流程已经设置过根,说明有代码路径重复调用,属于逻辑错误。
六、超级块的查找与创建:sget_fc 的并发控制
get_tree_bdev() 里的 sget_dev() 最终调用到 sget_fc():
c
struct super_block *sget_dev(struct fs_context *fc, dev_t dev)
{
fc->sget_key = &dev;
return sget_fc(fc, super_s_dev_test, super_s_dev_set);
}
sget_fc() 是超级块生命周期管理的核心,它要解决的核心问题是:多个进程同时挂载同一设备时,如何保证只创建一个超级块?
c
struct super_block *sget_fc(struct fs_context *fc,
int (*test)(struct super_block *, struct fs_context *),
int (*set)(struct super_block *, struct fs_context *))
{
struct super_block *s = NULL;
struct super_block *old;
struct user_namespace *user_ns = fc->global ? &init_user_ns : fc->user_ns;
int err;
retry:
spin_lock(&sb_lock);
if (test) {
hlist_for_each_entry(old, &fc->fs_type->fs_supers, s_instances) {
if (test(old, fc))
goto share_extant_sb;
}
}
if (!s) {
spin_unlock(&sb_lock);
s = alloc_super(fc->fs_type, fc->sb_flags, user_ns);
if (!s)
return ERR_PTR(-ENOMEM);
goto retry; // 分配后重试,重新查找
}
s->s_fs_info = fc->s_fs_info;
err = set(s, fc);
if (err) {
s->s_fs_info = NULL;
spin_unlock(&sb_lock);
destroy_unused_super(s);
return ERR_PTR(err);
}
fc->s_fs_info = NULL;
s->s_type = fc->fs_type;
s->s_iflags |= fc->s_iflags;
strscpy(s->s_id, s->s_type->name, sizeof(s->s_id));
/*
* Make the superblock visible on @super_blocks and @fs_supers.
* It's in a nascent state and users should wait on SB_BORN or
* SB_DYING to be set.
*/
list_add_tail(&s->s_list, &super_blocks);
hlist_add_head(&s->s_instances, &s->s_type->fs_supers);
spin_unlock(&sb_lock);
get_filesystem(s->s_type);
shrinker_register(s->s_shrink);
return s;
share_extant_sb:
if (user_ns != old->s_user_ns || fc->exclusive) {
spin_unlock(&sb_lock);
destroy_unused_super(s);
if (fc->exclusive)
warnfc(fc, "reusing existing filesystem not allowed");
else
warnfc(fc, "reusing existing filesystem in another namespace not allowed");
return ERR_PTR(-EBUSY);
}
if (!grab_super(old))
goto retry;
destroy_unused_super(s);
return old;
}
这段代码有几个精妙之处:
重试机制(goto retry) :sb_lock 是自旋锁,不能睡眠。所以 alloc_super() 必须在释放锁之后调用。但释放锁后,其它 CPU 可能已经创建了同样的超级块,所以分配完必须重新加锁、重新查找。这种"先释放锁分配,再重试查找"的模式是内核里处理"锁内不能睡眠"场景的经典范式。
test 与 set 回调 :test 用于判断两个超级块是否"相同"(对块设备文件系统而言是比较设备号),set 用于在新建超级块时写入关键字段(比如 s_dev)。这两个回调抽象了不同文件系统的"超级块身份判定"逻辑。
命名空间与独占检查 :share_extant_sb 标签下的逻辑很关键------如果现有超级块的 s_user_ns 与当前上下文的 user_ns 不同,或者上下文要求 exclusive(独占,不允许复用),就返回 -EBUSY。这是容器时代挂载隔离的核心机制------不允许跨用户命名空间共享超级块,否则容器 A 就能通过重新挂载看到容器 B 的文件系统。
状态可见性 :新建的超级块通过 list_add_tail() 和 hlist_add_head() 分别加入 super_blocks 全局链表与 fs_type->fs_supers 哈希链。注释里强调它处于"nascent(初生)状态",其它代码应当等待 SB_BORN 或 SB_DYING。这解释了为什么 vfs_get_tree() 要在 get_tree 成功后才调用 super_wake(sb, SB_BORN)------顺序不能颠倒。
七、ext4 的超级块填充:从 ext4_fill_super 到 __ext4_fill_super
超级块新建后,get_tree_bdev() 调用 fill_super(),即 ext4_fill_super():
c
static int ext4_fill_super(struct super_block *sb, struct fs_context *fc)
{
struct ext4_fs_context *ctx = fc->fs_private;
struct ext4_sb_info *sbi;
const char *descr;
int ret;
sbi = ext4_alloc_sbi(sb); // 分配 ext4 私有超级块信息
if (!sbi)
return -ENOMEM;
fc->s_fs_info = sbi;
strreplace(sb->s_id, '/', '!'); // 清理名字中的 '/'
sbi->s_sb_block = 1; /* Default super block location */
if (ctx->spec & EXT4_SPEC_s_sb_block)
sbi->s_sb_block = ctx->s_sb_block;
ret = __ext4_fill_super(fc, sb); // 【核心】实际的填充逻辑
//yym-gaizao
pr_info("YYM-DBG: [ext4_fill_super] Attempting to fill ext4 superblock for device %s \n", sb->s_id);
if (ret < 0)
goto free_sbi;
if (sbi->s_journal) {
if (test_opt(sb, DATA_FLAGS) == EXT4_MOUNT_JOURNAL_DATA)
descr = " journalled data mode";
else if (test_opt(sb, DATA_FLAGS) == EXT4_MOUNT_ORDERED_DATA)
descr = " ordered data mode";
else
descr = " writeback data mode";
} else
descr = "out journal";
if (___ratelimit(&ext4_mount_msg_ratelimit, "EXT4-fs mount"))
ext4_msg(sb, KERN_INFO, "mounted filesystem %pU %s with%s. "
"Quota mode: %s.", &sb->s_uuid,
sb_rdonly(sb) ? "ro" : "r/w", descr,
ext4_quota_mode(sb));
/* Update the s_overhead_clusters if necessary */
ext4_update_overhead(sb, false);
return 0;
free_sbi:
ext4_free_sbi(sbi);
fc->s_fs_info = NULL;
return ret;
}
ext4_fill_super() 本身只做"外壳"工作------分配 sbi、绑定到上下文、调用 __ext4_fill_super()、打印挂载信息。真正的重头戏在 __ext4_fill_super() 里,它有超过 400 行代码,是 ext4 挂载中最复杂、最关键的部分。挑出主线步骤:
c
static int __ext4_fill_super(struct fs_context *fc, struct super_block *sb)
{
...
/* 1. 读取磁盘超级块 */
err = ext4_load_super(sb, &logical_sb_block, silent);
if (err)
goto out_fail;
/* 2. 初始化元数据校验和 */
err = ext4_init_metadata_csum(sb, es);
if (err)
goto failed_mount;
/* 3. 设置默认挂载选项 */
ext4_set_def_opts(sb, es);
/* 4. 初始化 inode 相关信息 */
err = ext4_inode_info_init(sb, es);
if (err)
goto failed_mount;
/* 5. 应用超级块中的挂载选项 */
err = parse_apply_sb_mount_options(sb, ctx);
if (err < 0)
goto failed_mount;
/* 6. 检查选项一致性并应用 */
err = ext4_check_opt_consistency(fc, sb);
if (err < 0)
goto failed_mount;
ext4_apply_options(fc, sb);
/* 7. 初始化编码、检查特性兼容性 */
err = ext4_encoding_init(sb, es);
if (err)
goto failed_mount;
err = ext4_check_feature_compatibility(sb, es, silent);
if (err)
goto failed_mount;
/* 8. 初始化块组元数据、几何信息 */
err = ext4_block_group_meta_init(sb, silent);
if (err)
goto failed_mount;
err = ext4_group_desc_init(sb, es, logical_sb_block, &first_not_zeroed);
if (err)
goto failed_mount3;
/* 9. 注册 extent 状态收缩器 */
err = ext4_es_register_shrinker(sbi);
if (err)
goto failed_mount3;
/* 10. 加载并初始化日志 */
if (!test_opt(sb, NOLOAD) && ext4_has_feature_journal(sb)) {
err = ext4_load_and_init_journal(sb, es, ctx);
if (err)
goto failed_mount3a;
}
...
/* 11. 获取根 inode,创建根 dentry */
root = ext4_iget(sb, EXT4_ROOT_INO, EXT4_IGET_SPECIAL);
if (IS_ERR(root)) {
ext4_msg(sb, KERN_ERR, "get root inode failed");
err = PTR_ERR(root);
root = NULL;
goto failed_mount4;
}
sb->s_root = d_make_root(root); // 【关键】建立超级块根
if (!sb->s_root) {
ext4_msg(sb, KERN_ERR, "get root dentry failed");
err = -ENOMEM;
goto failed_mount4;
}
/* 12. 初始化 mballoc(多块分配器) */
err = ext4_mb_init(sb);
if (err) {
ext4_msg(sb, KERN_ERR, "failed to initialize mballoc (%d)", err);
goto failed_mount5;
}
/* 13. 注册 sysfs、初始化孤儿信息 */
err = ext4_register_sysfs(sb);
if (err)
goto failed_mount7;
err = ext4_init_orphan_info(sb);
if (err)
goto failed_mount8;
/* 14. 清理孤儿 inode,标记恢复完成 */
ext4_orphan_cleanup(sb, es);
ext4_superblock_csum_set(sb);
if (needs_recovery) {
ext4_msg(sb, KERN_INFO, "recovery complete");
err = ext4_mark_recovery_complete(sb, es);
if (err)
goto failed_mount10;
}
return 0;
...
}
这段代码覆盖了 ext4 挂载的所有关键阶段:读取超级块、校验元数据、解析选项、初始化块组、加载日志、获取根 inode、建立 s_root、初始化分配器、注册 sysfs、清理孤儿 inode、标记恢复完成。
特别注意 d_make_root(root) 这一行 :它把根 inode 包装成根 dentry,赋给 sb->s_root。这正是上游 vfs_get_tree() 里 BUG_ON(!fc->root) 所期待的最终结果------通过 get_tree_bdev() 里的 fc->root = dget(s->s_root) 传回上下文。
错误处理:goto 链的设计
__ext4_fill_super() 的尾部是一长串 failed_mount* 标签,这是内核错误处理的经典范式:
c
failed_mount10:
ext4_quotas_off(sb, EXT4_MAXQUOTAS);
failed_mount9: __maybe_unused
ext4_release_orphan_info(sb);
failed_mount8:
ext4_unregister_sysfs(sb);
kobject_put(&sbi->s_kobj);
failed_mount7:
ext4_unregister_li_request(sb);
failed_mount6:
ext4_mb_release(sb);
ext4_flex_groups_free(sbi);
ext4_percpu_param_destroy(sbi);
failed_mount5:
ext4_ext_release(sb);
ext4_release_system_zone(sb);
failed_mount4a:
dput(sb->s_root);
sb->s_root = NULL;
failed_mount4:
ext4_msg(sb, KERN_ERR, "mount failed");
if (EXT4_SB(sb)->rsv_conversion_wq)
destroy_workqueue(EXT4_SB(sb)->rsv_conversion_wq);
failed_mount_wq:
ext4_xattr_destroy_cache(sbi->s_ea_inode_cache);
sbi->s_ea_inode_cache = NULL;
ext4_xattr_destroy_cache(sbi->s_ea_block_cache);
sbi->s_ea_block_cache = NULL;
if (sbi->s_journal) {
flush_work(&sbi->s_sb_upd_work);
jbd2_journal_destroy(sbi->s_journal);
sbi->s_journal = NULL;
}
...
out_fail:
invalidate_bdev(sb->s_bdev);
sb->s_fs_info = NULL;
return err;
}
这种"倒序释放"的设计非常精妙:每个错误分支跳转到对应的标签,从该标签开始依次执行所有后续清理 ,直到函数返回。这样既避免了代码重复,又保证了资源按获取的逆序释放。阅读内核代码时,看错误处理标签的排布,就能大致推断出资源获取的顺序和模块间的依赖关系 ------例如 failed_mount6 在 failed_mount5 之前,说明 ext4_mb_init()(对应 failed_mount6)在 ext4_ext_release()(对应 failed_mount5)之前执行。
八、回顾完整调用链
把上面所有环节串起来,一次挂载的完整路径是:
text
用户空间 mount("/dev/sda1", "/mnt", "ext4", 0, "data=ordered")
│
▼
SYSCALL_DEFINE5(mount, ...)
│
▼
path_mount() → do_new_mount()
├─ get_fs_type("ext4") // 查找 file_system_type,必要时加载模块
├─ fs_context_for_mount(type, sb_flags)
│ └─ alloc_fs_context(..., FS_CONTEXT_FOR_MOUNT)
│ ├─ kzalloc(struct fs_context)
│ ├─ get_filesystem / get_current_cred / get_net / get_user_ns
│ └─ ext4_init_fs_context()
│ ├─ kzalloc(struct ext4_fs_context)
│ └─ fc->ops = &ext4_context_ops
├─ vfs_parse_fs_string(fc, "source", ...) // 解析 source
├─ parse_monolithic_mount_data(fc, data) // 解析 data 字符串
├─ vfs_get_tree(fc)
│ └─ fc->ops->get_tree(fc) // 回调到 ext4
│ └─ ext4_get_tree(fc)
│ └─ get_tree_bdev(fc, ext4_fill_super)
│ ├─ lookup_bdev(fc->source, &dev)
│ ├─ sget_dev(fc, dev)
│ │ └─ sget_fc(fc, super_s_dev_test, super_s_dev_set)
│ │ ├─ 遍历 fs_supers 查找现有超级块
│ │ ├─ 若无,alloc_super() 分配
│ │ ├─ list_add_tail / hlist_add_head 发布
│ │ └─ grab_super 或返回新建的 s
│ ├─ setup_bdev_super(s, flags, fc)
│ ├─ fill_super(s, fc) // 即 ext4_fill_super
│ │ ├─ ext4_alloc_sbi(sb)
│ │ └─ __ext4_fill_super(fc, sb)
│ │ ├─ ext4_load_super
│ │ ├─ ext4_init_metadata_csum
│ │ ├─ ext4_apply_options
│ │ ├─ ext4_load_and_init_journal
│ │ ├─ ext4_iget(EXT4_ROOT_INO)
│ │ ├─ d_make_root(root) → sb->s_root
│ │ ├─ ext4_mb_init
│ │ ├─ ext4_register_sysfs
│ │ └─ ext4_orphan_cleanup
│ └─ fc->root = dget(s->s_root) // 交回上下文
├─ super_wake(sb, SB_BORN) // 标记超级块就绪
├─ security_sb_set_mnt_opts(sb, ...) // LSM 检查
└─ do_new_mount_fc(fc, path, mnt_flags) // 把挂载点加入命名空间
九、设计哲学的思考
读完整条链,我们可以总结出内核挂载框架的几个设计哲学:
第一,抽象分层清晰。 do_new_mount() 只负责"通用流程",fs_context 负责"配置传递",file_system_type 负责"类型描述",super_block 负责"运行时状态"。每一层各司其职,不越界。
第二,回调无处不在。 init_fs_context、parse_param、get_tree、reconfigure、free、test、set------这些回调把"通用框架"与"文件系统特定逻辑"解耦,使 ext4、xfs、btrfs 能共享大部分挂载代码。
第三,错误处理即文档。 goto 标签的排布顺序反映了资源获取顺序;BUG() 与 WARN() 的位置反映了接口契约;注释里的"why"比"what"更被重视。
第四,并发安全优先。 sget_fc() 的"释放锁-分配-重试"模式、sb_lock 短临界区、try_module_get 引用计数,都体现了内核对并发场景的严谨态度。
第五,兼容性渐进。 fc->oldapi 标记、legacy_init_fs_context() 回退、parse_monolithic_mount_data() 兼容传统字符串------新框架在引入的同时,为旧文件系统保留了完整的生存空间。
理解这条调用链,不仅能帮助我们分析挂载失败、死锁、资源泄漏等实际问题的根因,更能让我们体会内核工程师如何在"性能、正确性、可维护性"三者之间做出精妙的权衡。下一次你在代码里加一行调试打印,不妨先想一想:这行代码处于调用链的哪一环?它持有的锁有哪些?它依赖的子系统是否已经就绪?------这些问题的答案,往往就藏在本文剖析的这些函数里。
#源码
cpp
/*
* create a new mount for userspace and request it to be added into the
* namespace's tree
*/
static int do_new_mount(struct path *path, const char *fstype, int sb_flags,
int mnt_flags, const char *name, void *data)
{
struct file_system_type *type;
struct fs_context *fc;
const char *subtype = NULL;
int err = 0;
if (!fstype)
return -EINVAL;
type = get_fs_type(fstype);
if (!type)
return -ENODEV;
if (type->fs_flags & FS_HAS_SUBTYPE) {
subtype = strchr(fstype, '.');
if (subtype) {
subtype++;
if (!*subtype) {
put_filesystem(type);
return -EINVAL;
}
}
}
fc = fs_context_for_mount(type, sb_flags);
put_filesystem(type);
if (IS_ERR(fc))
return PTR_ERR(fc);
/*
* Indicate to the filesystem that the mount request is coming
* from the legacy mount system call.
*/
fc->oldapi = true;
if (subtype)
err = vfs_parse_fs_string(fc, "subtype",
subtype, strlen(subtype));
if (!err && name)
err = vfs_parse_fs_string(fc, "source", name, strlen(name));
if (!err)
err = parse_monolithic_mount_data(fc, data);
if (!err && !mount_capable(fc))
err = -EPERM;
if (!err)
err = vfs_get_tree(fc);
if (!err)
err = do_new_mount_fc(fc, path, mnt_flags);
put_fs_context(fc);
return err;
}
struct fs_context *fs_context_for_mount(struct file_system_type *fs_type,
unsigned int sb_flags)
{
return alloc_fs_context(fs_type, NULL, sb_flags, 0,
FS_CONTEXT_FOR_MOUNT);
}
EXPORT_SYMBOL(fs_context_for_mount);
/**
* alloc_fs_context - Create a filesystem context.
* @fs_type: The filesystem type.
* @reference: The dentry from which this one derives (or NULL)
* @sb_flags: Filesystem/superblock flags (SB_*)
* @sb_flags_mask: Applicable members of @sb_flags
* @purpose: The purpose that this configuration shall be used for.
*
* Open a filesystem and create a mount context. The mount context is
* initialised with the supplied flags and, if a submount/automount from
* another superblock (referred to by @reference) is supplied, may have
* parameters such as namespaces copied across from that superblock.
*/
static struct fs_context *alloc_fs_context(struct file_system_type *fs_type,
struct dentry *reference,
unsigned int sb_flags,
unsigned int sb_flags_mask,
enum fs_context_purpose purpose)
{
int (*init_fs_context)(struct fs_context *);
struct fs_context *fc;
int ret = -ENOMEM;
fc = kzalloc(sizeof(struct fs_context), GFP_KERNEL_ACCOUNT);
if (!fc)
return ERR_PTR(-ENOMEM);
fc->purpose = purpose;
fc->sb_flags = sb_flags;
fc->sb_flags_mask = sb_flags_mask;
fc->fs_type = get_filesystem(fs_type);
fc->cred = get_current_cred();
fc->net_ns = get_net(current->nsproxy->net_ns);
fc->log.prefix = fs_type->name;
mutex_init(&fc->uapi_mutex);
switch (purpose) {
case FS_CONTEXT_FOR_MOUNT:
fc->user_ns = get_user_ns(fc->cred->user_ns);
break;
case FS_CONTEXT_FOR_SUBMOUNT:
fc->user_ns = get_user_ns(reference->d_sb->s_user_ns);
break;
case FS_CONTEXT_FOR_RECONFIGURE:
atomic_inc(&reference->d_sb->s_active);
fc->user_ns = get_user_ns(reference->d_sb->s_user_ns);
fc->root = dget(reference);
break;
}
/* TODO: Make all filesystems support this unconditionally */
init_fs_context = fc->fs_type->init_fs_context;
if (!init_fs_context)
init_fs_context = legacy_init_fs_context;
ret = init_fs_context(fc);
if (ret < 0)
goto err_fc;
fc->need_free = true;
return fc;
err_fc:
put_fs_context(fc);
return ERR_PTR(ret);
}
struct file_system_type *get_fs_type(const char *name)
{
struct file_system_type *fs;
const char *dot = strchr(name, '.');
int len = dot ? dot - name : strlen(name);
fs = __get_fs_type(name, len);
if (!fs && (request_module("fs-%.*s", len, name) == 0)) {
fs = __get_fs_type(name, len);
if (!fs)
pr_warn_once("request_module fs-%.*s succeeded, but still no fs?\n",
len, name);
}
if (dot && fs && !(fs->fs_flags & FS_HAS_SUBTYPE)) {
put_filesystem(fs);
fs = NULL;
}
return fs;
}
EXPORT_SYMBOL(get_fs_type);
static struct file_system_type *__get_fs_type(const char *name, int len)
{
struct file_system_type *fs;
read_lock(&file_systems_lock);
fs = *(find_filesystem(name, len));
if (fs && !try_module_get(fs->owner))
fs = NULL;
read_unlock(&file_systems_lock);
return fs;
}
static struct file_system_type **find_filesystem(const char *name, unsigned len)
{
struct file_system_type **p;
for (p = &file_systems; *p; p = &(*p)->next)
if (strncmp((*p)->name, name, len) == 0 &&
!(*p)->name[len])
break;
return p;
}
/**
* vfs_get_tree - Get the mountable root
* @fc: The superblock configuration context.
*
* The filesystem is invoked to get or create a superblock which can then later
* be used for mounting. The filesystem places a pointer to the root to be
* used for mounting in @fc->root.
*/
int vfs_get_tree(struct fs_context *fc)
{
struct super_block *sb;
int error;
if (fc->root)
return -EBUSY;
/* Get the mountable root in fc->root, with a ref on the root and a ref
* on the superblock.
*/
error = fc->ops->get_tree(fc);
if (error < 0)
return error;
if (!fc->root) {
pr_err("Filesystem %s get_tree() didn't set fc->root\n",
fc->fs_type->name);
/* We don't know what the locking state of the superblock is -
* if there is a superblock.
*/
BUG();
}
sb = fc->root->d_sb;
WARN_ON(!sb->s_bdi);
/*
* super_wake() contains a memory barrier which also care of
* ordering for super_cache_count(). We place it before setting
* SB_BORN as the data dependency between the two functions is
* the superblock structure contents that we just set up, not
* the SB_BORN flag.
*/
super_wake(sb, SB_BORN);
error = security_sb_set_mnt_opts(sb, fc->security, 0, NULL);
if (unlikely(error)) {
fc_drop_locked(fc);
return error;
}
/*
* filesystems should never set s_maxbytes larger than MAX_LFS_FILESIZE
* but s_maxbytes was an unsigned long long for many releases. Throw
* this warning for a little while to try and catch filesystems that
* violate this rule.
*/
WARN((sb->s_maxbytes < 0), "%s set sb->s_maxbytes to "
"negative value (%lld)\n", fc->fs_type->name, sb->s_maxbytes);
return 0;
}
EXPORT_SYMBOL(vfs_get_tree);
static int ext4_get_tree(struct fs_context *fc)
{
return get_tree_bdev(fc, ext4_fill_super);
}
static int __ext4_fill_super(struct fs_context *fc, struct super_block *sb)
{
struct ext4_super_block *es = NULL;
struct ext4_sb_info *sbi = EXT4_SB(sb);
ext4_fsblk_t logical_sb_block;
struct inode *root;
int needs_recovery;
int err;
ext4_group_t first_not_zeroed;
struct ext4_fs_context *ctx = fc->fs_private;
int silent = fc->sb_flags & SB_SILENT;
/* Set defaults for the variables that will be set during parsing */
if (!(ctx->spec & EXT4_SPEC_JOURNAL_IOPRIO))
ctx->journal_ioprio = DEFAULT_JOURNAL_IOPRIO;
sbi->s_inode_readahead_blks = EXT4_DEF_INODE_READAHEAD_BLKS;
sbi->s_sectors_written_start =
part_stat_read(sb->s_bdev, sectors[STAT_WRITE]);
err = ext4_load_super(sb, &logical_sb_block, silent);
if (err)
goto out_fail;
es = sbi->s_es;
sbi->s_kbytes_written = le64_to_cpu(es->s_kbytes_written);
err = ext4_init_metadata_csum(sb, es);
if (err)
goto failed_mount;
ext4_set_def_opts(sb, es);
sbi->s_resuid = make_kuid(&init_user_ns, le16_to_cpu(es->s_def_resuid));
sbi->s_resgid = make_kgid(&init_user_ns, le16_to_cpu(es->s_def_resgid));
sbi->s_commit_interval = JBD2_DEFAULT_MAX_COMMIT_AGE * HZ;
sbi->s_min_batch_time = EXT4_DEF_MIN_BATCH_TIME;
sbi->s_max_batch_time = EXT4_DEF_MAX_BATCH_TIME;
/*
* set default s_li_wait_mult for lazyinit, for the case there is
* no mount option specified.
*/
sbi->s_li_wait_mult = EXT4_DEF_LI_WAIT_MULT;
err = ext4_inode_info_init(sb, es);
if (err)
goto failed_mount;
err = parse_apply_sb_mount_options(sb, ctx);
if (err < 0)
goto failed_mount;
sbi->s_def_mount_opt = sbi->s_mount_opt;
sbi->s_def_mount_opt2 = sbi->s_mount_opt2;
err = ext4_check_opt_consistency(fc, sb);
if (err < 0)
goto failed_mount;
ext4_apply_options(fc, sb);
err = ext4_encoding_init(sb, es);
if (err)
goto failed_mount;
err = ext4_check_journal_data_mode(sb);
if (err)
goto failed_mount;
sb->s_flags = (sb->s_flags & ~SB_POSIXACL) |
(test_opt(sb, POSIX_ACL) ? SB_POSIXACL : 0);
/* i_version is always enabled now */
sb->s_flags |= SB_I_VERSION;
err = ext4_check_feature_compatibility(sb, es, silent);
if (err)
goto failed_mount;
err = ext4_block_group_meta_init(sb, silent);
if (err)
goto failed_mount;
ext4_hash_info_init(sb);
err = ext4_handle_clustersize(sb);
if (err)
goto failed_mount;
err = ext4_check_geometry(sb, es);
if (err)
goto failed_mount;
timer_setup(&sbi->s_err_report, print_daily_error_info, 0);
spin_lock_init(&sbi->s_error_lock);
INIT_WORK(&sbi->s_sb_upd_work, update_super_work);
err = ext4_group_desc_init(sb, es, logical_sb_block, &first_not_zeroed);
if (err)
goto failed_mount3;
err = ext4_es_register_shrinker(sbi);
if (err)
goto failed_mount3;
sbi->s_stripe = ext4_get_stripe_size(sbi);
/*
* It's hard to get stripe aligned blocks if stripe is not aligned with
* cluster, just disable stripe and alert user to simpfy code and avoid
* stripe aligned allocation which will rarely successes.
*/
if (sbi->s_stripe > 0 && sbi->s_cluster_ratio > 1 &&
sbi->s_stripe % sbi->s_cluster_ratio != 0) {
ext4_msg(sb, KERN_WARNING,
"stripe (%lu) is not aligned with cluster size (%u), "
"stripe is disabled",
sbi->s_stripe, sbi->s_cluster_ratio);
sbi->s_stripe = 0;
}
sbi->s_extent_max_zeroout_kb = 32;
/*
* set up enough so that it can read an inode
*/
sb->s_op = &ext4_sops;
sb->s_export_op = &ext4_export_ops;
sb->s_xattr = ext4_xattr_handlers;
#ifdef CONFIG_FS_ENCRYPTION
sb->s_cop = &ext4_cryptops;
#endif
#ifdef CONFIG_FS_VERITY
sb->s_vop = &ext4_verityops;
#endif
#ifdef CONFIG_QUOTA
sb->dq_op = &ext4_quota_operations;
if (ext4_has_feature_quota(sb))
sb->s_qcop = &dquot_quotactl_sysfile_ops;
else
sb->s_qcop = &ext4_qctl_operations;
sb->s_quota_types = QTYPE_MASK_USR | QTYPE_MASK_GRP | QTYPE_MASK_PRJ;
#endif
memcpy(&sb->s_uuid, es->s_uuid, sizeof(es->s_uuid));
INIT_LIST_HEAD(&sbi->s_orphan); /* unlinked but open files */
mutex_init(&sbi->s_orphan_lock);
ext4_fast_commit_init(sb);
sb->s_root = NULL;
needs_recovery = (es->s_last_orphan != 0 ||
ext4_has_feature_orphan_present(sb) ||
ext4_has_feature_journal_needs_recovery(sb));
if (ext4_has_feature_mmp(sb) && !sb_rdonly(sb)) {
err = ext4_multi_mount_protect(sb, le64_to_cpu(es->s_mmp_block));
if (err)
goto failed_mount3a;
}
err = -EINVAL;
/*
* The first inode we look at is the journal inode. Don't try
* root first: it may be modified in the journal!
*/
if (!test_opt(sb, NOLOAD) && ext4_has_feature_journal(sb)) {
err = ext4_load_and_init_journal(sb, es, ctx);
if (err)
goto failed_mount3a;
} else if (test_opt(sb, NOLOAD) && !sb_rdonly(sb) &&
ext4_has_feature_journal_needs_recovery(sb)) {
ext4_msg(sb, KERN_ERR, "required journal recovery "
"suppressed and not mounted read-only");
goto failed_mount3a;
} else {
/* Nojournal mode, all journal mount options are illegal */
if (test_opt(sb, JOURNAL_ASYNC_COMMIT)) {
ext4_msg(sb, KERN_ERR, "can't mount with "
"journal_async_commit, fs mounted w/o journal");
goto failed_mount3a;
}
if (test_opt2(sb, EXPLICIT_JOURNAL_CHECKSUM)) {
ext4_msg(sb, KERN_ERR, "can't mount with "
"journal_checksum, fs mounted w/o journal");
goto failed_mount3a;
}
if (sbi->s_commit_interval != JBD2_DEFAULT_MAX_COMMIT_AGE*HZ) {
ext4_msg(sb, KERN_ERR, "can't mount with "
"commit=%lu, fs mounted w/o journal",
sbi->s_commit_interval / HZ);
goto failed_mount3a;
}
if (EXT4_MOUNT_DATA_FLAGS &
(sbi->s_mount_opt ^ sbi->s_def_mount_opt)) {
ext4_msg(sb, KERN_ERR, "can't mount with "
"data=, fs mounted w/o journal");
goto failed_mount3a;
}
sbi->s_def_mount_opt &= ~EXT4_MOUNT_JOURNAL_CHECKSUM;
clear_opt(sb, JOURNAL_CHECKSUM);
clear_opt(sb, DATA_FLAGS);
clear_opt2(sb, JOURNAL_FAST_COMMIT);
sbi->s_journal = NULL;
needs_recovery = 0;
}
if (!test_opt(sb, NO_MBCACHE)) {
sbi->s_ea_block_cache = ext4_xattr_create_cache();
if (!sbi->s_ea_block_cache) {
ext4_msg(sb, KERN_ERR,
"Failed to create ea_block_cache");
err = -EINVAL;
goto failed_mount_wq;
}
if (ext4_has_feature_ea_inode(sb)) {
sbi->s_ea_inode_cache = ext4_xattr_create_cache();
if (!sbi->s_ea_inode_cache) {
ext4_msg(sb, KERN_ERR,
"Failed to create ea_inode_cache");
err = -EINVAL;
goto failed_mount_wq;
}
}
}
/*
* Get the # of file system overhead blocks from the
* superblock if present.
*/
sbi->s_overhead = le32_to_cpu(es->s_overhead_clusters);
/* ignore the precalculated value if it is ridiculous */
if (sbi->s_overhead > ext4_blocks_count(es))
sbi->s_overhead = 0;
/*
* If the bigalloc feature is not enabled recalculating the
* overhead doesn't take long, so we might as well just redo
* it to make sure we are using the correct value.
*/
if (!ext4_has_feature_bigalloc(sb))
sbi->s_overhead = 0;
if (sbi->s_overhead == 0) {
err = ext4_calculate_overhead(sb);
if (err)
goto failed_mount_wq;
}
/*
* The maximum number of concurrent works can be high and
* concurrency isn't really necessary. Limit it to 1.
*/
EXT4_SB(sb)->rsv_conversion_wq =
alloc_workqueue("ext4-rsv-conversion", WQ_MEM_RECLAIM | WQ_UNBOUND, 1);
if (!EXT4_SB(sb)->rsv_conversion_wq) {
printk(KERN_ERR "EXT4-fs: failed to create workqueue\n");
err = -ENOMEM;
goto failed_mount4;
}
/*
* The jbd2_journal_load will have done any necessary log recovery,
* so we can safely mount the rest of the filesystem now.
*/
root = ext4_iget(sb, EXT4_ROOT_INO, EXT4_IGET_SPECIAL);
if (IS_ERR(root)) {
ext4_msg(sb, KERN_ERR, "get root inode failed");
err = PTR_ERR(root);
root = NULL;
goto failed_mount4;
}
if (!S_ISDIR(root->i_mode) || !root->i_blocks || !root->i_size) {
ext4_msg(sb, KERN_ERR, "corrupt root inode, run e2fsck");
iput(root);
err = -EFSCORRUPTED;
goto failed_mount4;
}
sb->s_root = d_make_root(root);
if (!sb->s_root) {
ext4_msg(sb, KERN_ERR, "get root dentry failed");
err = -ENOMEM;
goto failed_mount4;
}
err = ext4_setup_super(sb, es, sb_rdonly(sb));
if (err == -EROFS) {
sb->s_flags |= SB_RDONLY;
} else if (err)
goto failed_mount4a;
ext4_set_resv_clusters(sb);
if (test_opt(sb, BLOCK_VALIDITY)) {
err = ext4_setup_system_zone(sb);
if (err) {
ext4_msg(sb, KERN_ERR, "failed to initialize system "
"zone (%d)", err);
goto failed_mount4a;
}
}
ext4_fc_replay_cleanup(sb);
ext4_ext_init(sb);
/*
* Enable optimize_scan if number of groups is > threshold. This can be
* turned off by passing "mb_optimize_scan=0". This can also be
* turned on forcefully by passing "mb_optimize_scan=1".
*/
if (!(ctx->spec & EXT4_SPEC_mb_optimize_scan)) {
if (sbi->s_groups_count >= MB_DEFAULT_LINEAR_SCAN_THRESHOLD)
set_opt2(sb, MB_OPTIMIZE_SCAN);
else
clear_opt2(sb, MB_OPTIMIZE_SCAN);
}
err = ext4_mb_init(sb);
if (err) {
ext4_msg(sb, KERN_ERR, "failed to initialize mballoc (%d)",
err);
goto failed_mount5;
}
/*
* We can only set up the journal commit callback once
* mballoc is initialized
*/
if (sbi->s_journal)
sbi->s_journal->j_commit_callback =
ext4_journal_commit_callback;
err = ext4_percpu_param_init(sbi);
if (err)
goto failed_mount6;
if (ext4_has_feature_flex_bg(sb))
if (!ext4_fill_flex_info(sb)) {
ext4_msg(sb, KERN_ERR,
"unable to initialize "
"flex_bg meta info!");
err = -ENOMEM;
goto failed_mount6;
}
err = ext4_register_li_request(sb, first_not_zeroed);
if (err)
goto failed_mount6;
err = ext4_register_sysfs(sb);
if (err)
goto failed_mount7;
err = ext4_init_orphan_info(sb);
if (err)
goto failed_mount8;
#ifdef CONFIG_QUOTA
/* Enable quota usage during mount. */
if (ext4_has_feature_quota(sb) && !sb_rdonly(sb)) {
err = ext4_enable_quotas(sb);
if (err)
goto failed_mount9;
}
#endif /* CONFIG_QUOTA */
/*
* Save the original bdev mapping's wb_err value which could be
* used to detect the metadata async write error.
*/
spin_lock_init(&sbi->s_bdev_wb_lock);
errseq_check_and_advance(&sb->s_bdev->bd_inode->i_mapping->wb_err,
&sbi->s_bdev_wb_err);
EXT4_SB(sb)->s_mount_state |= EXT4_ORPHAN_FS;
ext4_orphan_cleanup(sb, es);
EXT4_SB(sb)->s_mount_state &= ~EXT4_ORPHAN_FS;
/*
* Update the checksum after updating free space/inode counters and
* ext4_orphan_cleanup. Otherwise the superblock can have an incorrect
* checksum in the buffer cache until it is written out and
* e2fsprogs programs trying to open a file system immediately
* after it is mounted can fail.
*/
ext4_superblock_csum_set(sb);
if (needs_recovery) {
ext4_msg(sb, KERN_INFO, "recovery complete");
err = ext4_mark_recovery_complete(sb, es);
if (err)
goto failed_mount10;
}
if (test_opt(sb, DISCARD) && !bdev_max_discard_sectors(sb->s_bdev))
ext4_msg(sb, KERN_WARNING,
"mounting with \"discard\" option, but the device does not support discard");
if (es->s_error_count)
mod_timer(&sbi->s_err_report, jiffies + 300*HZ); /* 5 minutes */
/* Enable message ratelimiting. Default is 10 messages per 5 secs. */
ratelimit_state_init(&sbi->s_err_ratelimit_state, 5 * HZ, 10);
ratelimit_state_init(&sbi->s_warning_ratelimit_state, 5 * HZ, 10);
ratelimit_state_init(&sbi->s_msg_ratelimit_state, 5 * HZ, 10);
atomic_set(&sbi->s_warning_count, 0);
atomic_set(&sbi->s_msg_count, 0);
return 0;
failed_mount10:
ext4_quotas_off(sb, EXT4_MAXQUOTAS);
failed_mount9: __maybe_unused
ext4_release_orphan_info(sb);
failed_mount8:
ext4_unregister_sysfs(sb);
kobject_put(&sbi->s_kobj);
failed_mount7:
ext4_unregister_li_request(sb);
failed_mount6:
ext4_mb_release(sb);
ext4_flex_groups_free(sbi);
ext4_percpu_param_destroy(sbi);
failed_mount5:
ext4_ext_release(sb);
ext4_release_system_zone(sb);
failed_mount4a:
dput(sb->s_root);
sb->s_root = NULL;
failed_mount4:
ext4_msg(sb, KERN_ERR, "mount failed");
if (EXT4_SB(sb)->rsv_conversion_wq)
destroy_workqueue(EXT4_SB(sb)->rsv_conversion_wq);
failed_mount_wq:
ext4_xattr_destroy_cache(sbi->s_ea_inode_cache);
sbi->s_ea_inode_cache = NULL;
ext4_xattr_destroy_cache(sbi->s_ea_block_cache);
sbi->s_ea_block_cache = NULL;
if (sbi->s_journal) {
/* flush s_sb_upd_work before journal destroy. */
flush_work(&sbi->s_sb_upd_work);
jbd2_journal_destroy(sbi->s_journal);
sbi->s_journal = NULL;
}
failed_mount3a:
ext4_es_unregister_shrinker(sbi);
failed_mount3:
/* flush s_sb_upd_work before sbi destroy */
flush_work(&sbi->s_sb_upd_work);
del_timer_sync(&sbi->s_err_report);
ext4_stop_mmpd(sbi);
ext4_group_desc_free(sbi);
failed_mount:
if (sbi->s_chksum_driver)
crypto_free_shash(sbi->s_chksum_driver);
#if IS_ENABLED(CONFIG_UNICODE)
utf8_unload(sb->s_encoding);
#endif
#ifdef CONFIG_QUOTA
for (unsigned int i = 0; i < EXT4_MAXQUOTAS; i++)
kfree(get_qf_name(sb, sbi, i));
#endif
fscrypt_free_dummy_policy(&sbi->s_dummy_enc_policy);
brelse(sbi->s_sbh);
if (sbi->s_journal_bdev_handle) {
invalidate_bdev(sbi->s_journal_bdev_handle->bdev);
bdev_release(sbi->s_journal_bdev_handle);
}
out_fail:
invalidate_bdev(sb->s_bdev);
sb->s_fs_info = NULL;
return err;
}
static int ext4_fill_super(struct super_block *sb, struct fs_context *fc)
{
struct ext4_fs_context *ctx = fc->fs_private;
struct ext4_sb_info *sbi;
const char *descr;
int ret;
sbi = ext4_alloc_sbi(sb);
if (!sbi)
return -ENOMEM;
fc->s_fs_info = sbi;
/* Cleanup superblock name */
strreplace(sb->s_id, '/', '!');
sbi->s_sb_block = 1; /* Default super block location */
if (ctx->spec & EXT4_SPEC_s_sb_block)
sbi->s_sb_block = ctx->s_sb_block;
ret = __ext4_fill_super(fc, sb);
//yym-gaizao
pr_info("YYM-DBG: [ext4_fill_super] Attempting to fill ext4 superblock for device %s \n", sb->s_id);
if (ret < 0)
goto free_sbi;
if (sbi->s_journal) {
if (test_opt(sb, DATA_FLAGS) == EXT4_MOUNT_JOURNAL_DATA)
descr = " journalled data mode";
else if (test_opt(sb, DATA_FLAGS) == EXT4_MOUNT_ORDERED_DATA)
descr = " ordered data mode";
else
descr = " writeback data mode";
} else
descr = "out journal";
if (___ratelimit(&ext4_mount_msg_ratelimit, "EXT4-fs mount"))
ext4_msg(sb, KERN_INFO, "mounted filesystem %pU %s with%s. "
"Quota mode: %s.", &sb->s_uuid,
sb_rdonly(sb) ? "ro" : "r/w", descr,
ext4_quota_mode(sb));
/* Update the s_overhead_clusters if necessary */
ext4_update_overhead(sb, false);
return 0;
free_sbi:
ext4_free_sbi(sbi);
fc->s_fs_info = NULL;
return ret;
}
static const struct fs_context_operations ext4_context_ops = {
.parse_param = ext4_parse_param,
.get_tree = ext4_get_tree,
.reconfigure = ext4_reconfigure,
.free = ext4_fc_free,
};
int ext4_init_fs_context(struct fs_context *fc)
{
struct ext4_fs_context *ctx;
ctx = kzalloc(sizeof(struct ext4_fs_context), GFP_KERNEL);
if (!ctx)
return -ENOMEM;
fc->fs_private = ctx;
fc->ops = &ext4_context_ops;
return 0;
}
/**
* get_tree_bdev - Get a superblock based on a single block device
* @fc: The filesystem context holding the parameters
* @fill_super: Helper to initialise a new superblock
*/
int get_tree_bdev(struct fs_context *fc,
int (*fill_super)(struct super_block *,
struct fs_context *))
{
struct super_block *s;
int error = 0;
dev_t dev;
if (!fc->source)
return invalf(fc, "No source specified");
error = lookup_bdev(fc->source, &dev);
if (error) {
errorf(fc, "%s: Can't lookup blockdev", fc->source);
return error;
}
fc->sb_flags |= SB_NOSEC;
s = sget_dev(fc, dev);
if (IS_ERR(s))
return PTR_ERR(s);
if (s->s_root) {
/* Don't summarily change the RO/RW state. */
if ((fc->sb_flags ^ s->s_flags) & SB_RDONLY) {
warnf(fc, "%pg: Can't mount, would change RO state", s->s_bdev);
deactivate_locked_super(s);
return -EBUSY;
}
} else {
error = setup_bdev_super(s, fc->sb_flags, fc);
if (!error)
error = fill_super(s, fc);
if (error) {
deactivate_locked_super(s);
return error;
}
s->s_flags |= SB_ACTIVE;
}
BUG_ON(fc->root);
fc->root = dget(s->s_root);
return 0;
}
EXPORT_SYMBOL(get_tree_bdev);
/**
* sget_dev - Find or create a superblock by device number
* @fc: Filesystem context.
* @dev: device number
*
* Find or create a superblock using the provided device number that
* will be stored in fc->sget_key.
*
* If an extant superblock is matched, then that will be returned with
* an elevated reference count that the caller must transfer or discard.
*
* If no match is made, a new superblock will be allocated and basic
* initialisation will be performed (s_type, s_fs_info, s_id, s_dev will
* be set). The superblock will be published and it will be returned in
* a partially constructed state with SB_BORN and SB_ACTIVE as yet
* unset.
*
* Return: an existing or newly created superblock on success, an error
* pointer on failure.
*/
struct super_block *sget_dev(struct fs_context *fc, dev_t dev)
{
fc->sget_key = &dev;
return sget_fc(fc, super_s_dev_test, super_s_dev_set);
}
EXPORT_SYMBOL(sget_dev);
/**
* sget_fc - Find or create a superblock
* @fc: Filesystem context.
* @test: Comparison callback
* @set: Setup callback
*
* Create a new superblock or find an existing one.
*
* The @test callback is used to find a matching existing superblock.
* Whether or not the requested parameters in @fc are taken into account
* is specific to the @test callback that is used. They may even be
* completely ignored.
*
* If an extant superblock is matched, it will be returned unless:
*
* (1) the namespace the filesystem context @fc and the extant
* superblock's namespace differ
*
* (2) the filesystem context @fc has requested that reusing an extant
* superblock is not allowed
*
* In both cases EBUSY will be returned.
*
* If no match is made, a new superblock will be allocated and basic
* initialisation will be performed (s_type, s_fs_info and s_id will be
* set and the @set callback will be invoked), the superblock will be
* published and it will be returned in a partially constructed state
* with SB_BORN and SB_ACTIVE as yet unset.
*
* Return: On success, an extant or newly created superblock is
* returned. On failure an error pointer is returned.
*/
struct super_block *sget_fc(struct fs_context *fc,
int (*test)(struct super_block *, struct fs_context *),
int (*set)(struct super_block *, struct fs_context *))
{
struct super_block *s = NULL;
struct super_block *old;
struct user_namespace *user_ns = fc->global ? &init_user_ns : fc->user_ns;
int err;
retry:
spin_lock(&sb_lock);
if (test) {
hlist_for_each_entry(old, &fc->fs_type->fs_supers, s_instances) {
if (test(old, fc))
goto share_extant_sb;
}
}
if (!s) {
spin_unlock(&sb_lock);
s = alloc_super(fc->fs_type, fc->sb_flags, user_ns);
if (!s)
return ERR_PTR(-ENOMEM);
goto retry;
}
s->s_fs_info = fc->s_fs_info;
err = set(s, fc);
if (err) {
s->s_fs_info = NULL;
spin_unlock(&sb_lock);
destroy_unused_super(s);
return ERR_PTR(err);
}
fc->s_fs_info = NULL;
s->s_type = fc->fs_type;
s->s_iflags |= fc->s_iflags;
strscpy(s->s_id, s->s_type->name, sizeof(s->s_id));
/*
* Make the superblock visible on @super_blocks and @fs_supers.
* It's in a nascent state and users should wait on SB_BORN or
* SB_DYING to be set.
*/
list_add_tail(&s->s_list, &super_blocks);
hlist_add_head(&s->s_instances, &s->s_type->fs_supers);
spin_unlock(&sb_lock);
get_filesystem(s->s_type);
shrinker_register(s->s_shrink);
return s;
share_extant_sb:
if (user_ns != old->s_user_ns || fc->exclusive) {
spin_unlock(&sb_lock);
destroy_unused_super(s);
if (fc->exclusive)
warnfc(fc, "reusing existing filesystem not allowed");
else
warnfc(fc, "reusing existing filesystem in another namespace not allowed");
return ERR_PTR(-EBUSY);
}
if (!grab_super(old))
goto retry;
destroy_unused_super(s);
return old;
}
EXPORT_SYMBOL(sget_fc);