python解决xml文件中存在中文文字的问题

如下

<?xml version="1.0" ?><annotation>
	<folder>250-499</folder>
	<filename>250.jpg</filename>
	<path>/home/pengdezhi/数据标注/250-499/250.jpg</path>
	<source>
		<database>Unknown</database>
	</source>
	<size>
		<width>1085</width>
		<height>612</height>
		<depth>3</depth>
	</size>
	<segmented>0</segmented>
	<object>
    <name>helmet_off</name>
		<pose>Unspecified</pose>
		<truncated>0</truncated>
		<difficult>0</difficult>
		<bndbox>
			<xmin>612</xmin>
			<ymin>37</ymin>
			<xmax>630</xmax>
			<ymax>55</ymax>
		</bndbox>
	</object>
</annotation>

解决方法

import os
import os.path
import xml.etree.ElementTree as ET


headstr = """\
<annotation>
    <folder>VOC</folder>
    <filename>%s</filename>
    <path>zhangyt</path>
    <source>
        <database>My Database</database>
    </source>
    <size>
        <width>%d</width>
        <height>%d</height>
        <depth>%d</depth>
    </size>
    <segmented>0</segmented>
"""
objstr = """\
    <object>
        <name>%s</name>
        <pose>Unspecified</pose>
        <truncated>0</truncated>
        <difficult>0</difficult>
        <bndbox>
            <xmin>%d</xmin>
            <ymin>%d</ymin>
            <xmax>%d</xmax>
            <ymax>%d</ymax>
        </bndbox>
    </object>
"""
 
tailstr = '''\
</annotation>
'''

def write_xml(anno_path,head, objs, tail):
    f = open(anno_path, "w")
    f.write(head)
    for obj in objs:
        f.write(objstr%(obj[0],obj[1],obj[2],obj[3],obj[4]))
    f.write(tail)


def convertxml(path_xml,files, save_xml):

	for xmlFile in files:
		dataset = []
		print(xmlFile)
		f = open(os.path.join(path_xml,xmlFile))
		xml_text = f.read()
		root = ET.fromstring(xml_text)
		f.close()

		image_name = xmlFile.replace("xml", "jpg")
		# name = root.find('path').text
		width = int(root.findtext("./size/width"))
		height = int(root.findtext("./size/height"))
		depth = int(root.findtext("./size/depth"))

		head=headstr % (image_name, width, height, depth)
 
		for obj in root.iter("object"):
			name = str(obj.findtext("name"))
			xmin = int(obj.findtext("bndbox/xmin"))
			ymin = int(obj.findtext("bndbox/ymin"))
			xmax = int(obj.findtext("bndbox/xmax"))
			ymax = int(obj.findtext("bndbox/ymax"))
 
			if xmax == xmin or ymax == ymin:
				print(xmlFile)
			dataset.append([name, xmin, ymin, xmax, ymax])
		tail = tailstr
		write_xml(os.path.join(save_xml, xmlFile),head, dataset, tail)


if __name__ == "__main__":
	path_xml = "Annotations/"
	save_xml =  "Annotations_name/"
	if not os.path.exists(save_xml):
		os.mkdir(save_xml)
	xmlFile = os.listdir(path_xml)
	convertxml(path_xml,xmlFile,save_xml)

 

  • 1
    点赞
  • 6
    收藏
    觉得还不错? 一键收藏
  • 1
    评论

“相关推荐”对你有帮助么?

  • 非常没帮助
  • 没帮助
  • 一般
  • 有帮助
  • 非常有帮助
提交
评论 1
添加红包

请填写红包祝福语或标题

红包个数最小为10个

红包金额最低5元

当前余额3.43前往充值 >
需支付:10.00
成就一亿技术人!
领取后你会自动成为博主和红包主的粉丝 规则
hope_wisdom
发出的红包
实付
使用余额支付
点击重新获取
扫码支付
钱包余额 0

抵扣说明:

1.余额是钱包充值的虚拟货币,按照1:1的比例进行支付金额的抵扣。
2.余额无法直接购买下载,可以购买VIP、付费专栏及课程。

余额充值