YOLOv5 分类模型 数据集加载

YOLOv5 分类模型 数据集加载

flyfish

数据集的加载 python实现,不使用torch库

简化实现

py 复制代码
import os
import os.path
from typing import Any, Callable, cast, Dict, List, Optional, Tuple, Union


class DatasetFolder:

    def __init__(
        self,
        root: str,

    ) -> None:
        self.root=root
        classes, class_to_idx = self.find_classes(self.root)
        samples = self.make_dataset(self.root, class_to_idx)
        
        self.classes = classes
        self.class_to_idx = class_to_idx
        self.samples = samples
        self.targets = [s[1] for s in samples]


    @staticmethod
    def make_dataset(
        directory: str,
        class_to_idx: Optional[Dict[str, int]] = None,

    ) -> List[Tuple[str, int]]:
 
        directory = os.path.expanduser(directory)

        if class_to_idx is None:
            _, class_to_idx = self.find_classes(directory)
        elif not class_to_idx:
            raise ValueError("'class_to_index' must have at least one entry to collect any samples.")



        instances = []
        available_classes = set()
        for target_class in sorted(class_to_idx.keys()):
            class_index = class_to_idx[target_class]
            target_dir = os.path.join(directory, target_class)
            if not os.path.isdir(target_dir):
                continue
            for root, _, fnames in sorted(os.walk(target_dir, followlinks=True)):
                for fname in sorted(fnames):
                    path = os.path.join(root, fname)
                    if 1:#验证:
                        item = path, class_index
                        instances.append(item)

                        if target_class not in available_classes:
                            available_classes.add(target_class)

        empty_classes = set(class_to_idx.keys()) - available_classes
        if empty_classes:
            msg = f"Found no valid file for the classes {', '.join(sorted(empty_classes))}. "


        return instances

    def find_classes(self, directory: str) -> Tuple[List[str], Dict[str, int]]:
 
        classes = sorted(entry.name for entry in os.scandir(directory) if entry.is_dir())
        if not classes:
            raise FileNotFoundError(f"Couldn't find any class folder in {directory}.")

        class_to_idx = {cls_name: i for i, cls_name in enumerate(classes)}
        return classes, class_to_idx


    def __getitem__(self, index: int) -> Tuple[Any, Any]:

        path, target = self.samples[index]
        sample = self.loader(path)


        return sample, target

    def __len__(self) -> int:
        return len(self.samples)





dataset =  DatasetFolder(root="/media/a/flyfish/test");

print(dataset)
print("dataset.targets:",dataset.targets)
print("dataset.classes:",dataset.classes)
print("samples:",dataset.samples)

find_classes 将标签索引和标签内容对应

0,1,2是标签索引
'n01440764', 'n01443537', 'n01484850'是类别名字也是文件夹名字

按照升序排序

复制代码
dataset.targets: [0, 0, 0, 0, 1, 1, 1, 1, 2, 2, 2, 2]
dataset.classes: ['n01440764', 'n01443537', 'n01484850']

样本中一个是图像文件的绝对路径,后面的是标签

复制代码
samples: [('/media/a/flyfish/test/n01440764/ILSVRC2012_val_00000293.JPEG', 0),
          ('/media/a/flyfish/test/n01440764/ILSVRC2012_val_00002138.JPEG', 0),
          ('/media/a/flyfish/test/n01440764/ILSVRC2012_val_00003014.JPEG', 0),
          ('/media/a/flyfish/test/n01440764/ILSVRC2012_val_00006697.JPEG', 0),
          ('/media/a/flyfish/test/n01443537/ILSVRC2012_val_00000236.JPEG', 1),
          ('/media/a/flyfish/test/n01443537/ILSVRC2012_val_00000262.JPEG', 1),
          ('/media/a/flyfish/test/n01443537/ILSVRC2012_val_00000307.JPEG', 1),
          ('/media/a/flyfish/test/n01443537/ILSVRC2012_val_00000994.JPEG', 1),
          ('/media/a/flyfish/test/n01484850/ILSVRC2012_val_00002338.JPEG', 2),
          ('/media/a/flyfish/test/n01484850/ILSVRC2012_val_00002752.JPEG', 2),
          ('/media/a/flyfish/test/n01484850/ILSVRC2012_val_00004311.JPEG', 2),
          ('/media/a/flyfish/test/n01484850/ILSVRC2012_val_00004329.JPEG', 2)]

可以功能丰富一些,例如检测文件的扩展名是否是支持的图像文件

py 复制代码
IMG_EXTENSIONS = (".jpg", ".jpeg", ".png", ".ppm", ".bmp", ".pgm", ".tif", ".tiff", ".webp")

def has_file_allowed_extension(filename: str, extensions: Union[str, Tuple[str, ...]]) -> bool:
    """检查文件是否为允许的扩展名
    """
    return filename.lower().endswith(extensions if isinstance(extensions, str) else tuple(extensions))

def is_image_file(filename: str) -> bool:
    return has_file_allowed_extension(filename, IMG_EXTENSIONS)

测试

py 复制代码
r=is_image_file("/media/a/flyfish/data/imagewoof/val/n02086240/1.jpeg");

print(r)#True

r=is_image_file("/media/a/flyfish/data/imagewoof/val/n02086240/1.txt");

print(r)#False
相关推荐
JicasdC123asd1 小时前
Converse2D频域卷积上采样改进YOLOv26图像重建与细节恢复能力
人工智能·yolo·目标跟踪
FL16238631291 小时前
基于yolov8+pyqt5实现的水尺图像识别与水深计算系统
开发语言·qt·yolo
jay神2 小时前
基于YOLOv8的传送带异物检测系统
人工智能·python·深度学习·yolo·可视化·计算机毕业设计
智驱力人工智能2 小时前
馆藏文物预防性保护依赖的图像分析技术 文物损害检测 文物破损检测 文物损害识别误报率优化方案 文物安全巡查AI系统案例 智慧文保AI监测
人工智能·算法·安全·yolo·边缘计算
智驱力人工智能4 小时前
一盔一带AI抓拍系统能否破解非机动车执法取证难 骑行未戴头盔检测 电动车未戴头盔智能监测 摩托车头盔佩戴AI识别系统 边缘计算实时处理
人工智能·算法·yolo·目标检测·边缘计算
jay神4 小时前
基于YOLOv8的无人机识别与检测系统
人工智能·深度学习·yolo·目标检测·毕业设计·无人机
小高求学之路5 小时前
计算机视觉、YOLO算法模型训练、无人机监测人员密集自动识别
算法·yolo·计算机视觉
duyinbi751721 小时前
ADown高效下采样改进YOLOv26目标检测性能提升
yolo·目标检测·目标跟踪
AidLux21 小时前
手机上AidLux2.1.0 运行模型广场的yolov8模型
yolo·智能手机
阿钱真强道1 天前
29 Python 聚类:什么是聚类?它和分类到底有什么区别?
python·分类·聚类·监督学习·无监督学习·层次聚类·聚类评估