xml转化为txt数据的脚本,为yolo提供训练

这里写自定义目录标题

xml转化为txt数据的脚本

代码如下:

bash 复制代码
import xml.etree.ElementTree as ET
import os, cv2
import numpy as np
from os import listdir
from os.path import join

classes = []


def convert(size, box):
    dw = 1. / (size[0])
    dh = 1. / (size[1])
    x = (box[0] + box[1]) / 2.0 - 1
    y = (box[2] + box[3]) / 2.0 - 1
    w = box[1] - box[0]
    h = box[3] - box[2]
    x = x * dw
    w = w * dw
    y = y * dh
    h = h * dh
    return (x, y, w, h)


def convert_annotation(xmlpath, xmlname):
    with open(xmlpath, "r", encoding='utf-8') as in_file:
        txtname = xmlname[:-4] + '.txt'
        txtfile = os.path.join(txtpath, txtname)
        tree = ET.parse(in_file)
        root = tree.getroot()
        filename = root.find('filename')
        img = cv2.imdecode(np.fromfile('{}/{}.{}'.format(imgpath, xmlname[:-4], postfix), np.uint8), cv2.IMREAD_COLOR)
        h, w = img.shape[:2]
        res = []
        for obj in root.iter('object'):
            cls = obj.find('name').text
            if cls not in classes:
                classes.append(cls)
            cls_id = classes.index(cls)
            xmlbox = obj.find('bndbox')
            b = (float(xmlbox.find('xmin').text), float(xmlbox.find('xmax').text), float(xmlbox.find('ymin').text),
                 float(xmlbox.find('ymax').text))
            bb = convert((w, h), b)
            res.append(str(cls_id) + " " + " ".join([str(a) for a in bb]))
        if len(res) != 0:
            with open(txtfile, 'w+') as f:
                f.write('\n'.join(res))


if __name__ == "__main__":
    postfix = 'jpg'
    imgpath = 'JPEGImages_Val'
    xmlpath = 'Annotations_Val'
    txtpath = 'labels_Val'

    if not os.path.exists(txtpath):
        os.makedirs(txtpath, exist_ok=True)

    list = os.listdir(xmlpath)
    error_file_list = []
    for i in range(0, len(list)):
        try:
            path = os.path.join(xmlpath, list[i])
            if ('.xml' in path) or ('.XML' in path):
                convert_annotation(path, list[i])
                print(f'file {list[i]} convert success.')
            else:
                print(f'file {list[i]} is not xml format.')
        except Exception as e:
            print(f'file {list[i]} convert error.')
            print(f'error message:\n{e}')
            error_file_list.append(list[i])
    print(f'this file convert failure\n{error_file_list}')
    print(f'Dataset Classes:{classes}')
相关推荐
梦帮科技9 分钟前
端侧编译原理:TVM / MLIR 计算图模式匹配、算子融合与显存生命周期复用
网络·人工智能·深度学习·神经网络·自然语言处理·cnn·mlir
糖炒狗子25 分钟前
NeurIPS 2025 最佳论文逐行拆解:一个 sigmoid 门控,让 Attention Sink 从 46.7% 掉到 4.8%
人工智能·深度学习
饼饼学习空间智能1 小时前
低空经济进入新兴支柱产业:物理AI如何重构低空智能监管系统?
大数据·深度学习·机器学习
小静AI工程实验室1 小时前
Python 爬虫解析 JSON-LD:多块 script、@graph 与坏数据的 9 个边界
爬虫·python·json
jason.zeng@15022072 小时前
(八)现有架构上新增一个通用Excel导出工具
python·架构·langchain·excel·llama
袁袁袁袁满2 小时前
AI Agent 如何“看懂“互联网?
爬虫·python·自动化·爬虫实战·多线程爬虫
泡海椒2 小时前
JQuick-Excel 实战:用 VALIDATION 建立可维护的 Excel 导入校验
开发语言·python·excel
Duang007_2 小时前
生产可观测性:从“系统慢“到“根因“的完整链路(Go / TypeScript)
后端·python·golang·typescript·prometheus
jason.zeng@15022072 小时前
(九)多轮对话式新增维修记录实现方案
python·prompt·交互·llama
黑妹天下第一乖2 小时前
小智改造实战解读-首 token 延迟去哪了:云端与端侧大模型的分段对照
开发语言·人工智能·python·嵌入式硬件·自然语言处理·iot