2 N! N! c, O! h( u! w贴标过程可以随时返回修改,保存的文件会覆盖上一个。/ {3 P0 Y, a, t; ?- U" R( {3 x
E* d8 _( U F, k; H# S# j # K+ l; ]6 \; B& ~' ?% W完成注解后,打开XML文件,发现和PASCAL VOC格式一样。 9 U* w9 I$ V+ V& \. z 0 k- ^; x. H q& @% P & V. s+ T) K. \9 S; }将xml文件提取图像信息 7 H X4 L; X- F( @. Z( J* o下面列举如何将xml文件提取图像信息,图片保存到image文件夹,xml保存标注内容。图片和标注的文件名字一样的。2 i, U+ v7 H7 S# Z) f
( ?! A3 \; @" \; T- ?2 S% [- n# g/ Q3 t- W5 p
$ r5 n1 b4 s0 J% f- u+ ]: f9 Q1 o 2 V. ^2 y% d, x+ ~& Q, M3 `3 o下面是images图片中的一个。 , B# K& g2 Y- x# h* u0 I; Q 1 e# s" B, Y$ ?0 w3 J8 S, J9 H3 F/ y1 L- \( ^) ]
下面是对应的xml文件。, X% ?- X! ] b4 ~
+ J$ _1 u0 b2 e
" |. M1 o0 x+ S. |: K$ e
<annotation> i0 m4 S1 R* X7 k9 v" \3 E
<folder>train</folder>! v) A' q2 V8 N1 R
<filename>apple_30.jpg</filename> 0 e9 p r; A* u! Z+ E l <path>C:\tensorflow1\models\research\object_detection\images\train\apple_30.jpg</path> ) t4 \" z* [* ]0 C. d <source> 4 y& i. [- Z4 z; A+ W <database>Unknown</database> ' Y, ]+ Z: ?) j( d) c/ b% g' u2 L </source> + R, _9 G* N* H/ C5 u( L* M <size>) K6 H" m; K5 d) ?$ i; i
<width>800</width>/ g8 I, f% [2 w: X
<height>800</height> ' ]; _2 z* w x* s4 E8 ~ <depth>3</depth> 1 o8 p0 f6 |* ?2 {/ A </size> . D6 {$ v3 m3 n: S, z8 r. ~" \5 B <segmented>0</segmented> ) C& Q Y4 S) V. U2 l5 Y <object> ( I1 _6 D: Q* L) E: q8 x <name>apple</name> B# o1 e! r& B7 z( | <pose>Unspecified</pose> ) R' O2 ?/ ~. t- e3 G <truncated>0</truncated> + I4 t. P4 t- H: g( R, C1 { <difficult>0</difficult> / ]4 s `5 {- b( T" p2 n/ I! O <bndbox> r+ c( K) ^) ~ <xmin>254</xmin> % y! ?: X; `6 i) y' b <ymin>163</ymin>: g- ~) u. l9 k, ~+ A
<xmax>582</xmax> 9 L% f- I* y! V: {) V5 f; m/ A <ymax>487</ymax> ' C! H4 Q/ @# e8 _% m. K: S </bndbox>; p1 S, Y3 a8 Y6 z0 V; X
</object> 7 M. L; a" K' _# T: S6 g <object>4 N2 M) e, y8 T# w6 [
<name>apple</name>. `8 R# }- R) N, t; g6 i) D$ o
<pose>Unspecified</pose> m1 b; U4 T& T <truncated>0</truncated> . @ [4 j" g3 o <difficult>0</difficult>4 z$ W$ y3 I5 T' f$ z) k
<bndbox> 8 c* _ t- T6 w/ y4 H7 z <xmin>217</xmin>/ D P" R# H7 d: q! ~
<ymin>448</ymin> . u6 g/ `( X( Q2 O <xmax>535</xmax>6 l6 J& _8 }* U" m7 l+ K7 i; z
<ymax>713</ymax>, V3 @8 m+ O( a& [. i, X
</bndbox>) x. z4 `/ C i
</object> 0 G n: `3 p/ q! F <object> . R* F, ~& V! |, o( |1 B <name>apple</name> 5 M: |2 E# p, d1 @1 {, Q <pose>Unspecified</pose>3 g; Q# k' W* D) K. d* m" \
<truncated>1</truncated> ; n; [5 ~9 {; U <difficult>0</difficult> $ v, N1 ^) P. q' D) ]4 ^/ F <bndbox> & o1 h, g) B$ @# p2 k <xmin>603</xmin>- R9 i: z& n9 s
<ymin>470</ymin>: ]2 f9 w$ r+ G; f$ T6 x
<xmax>800</xmax> : I% v2 e2 u! g <ymax>716</ymax> / F1 @3 Y/ X4 L5 P0 K- b4 Y </bndbox> , V1 ]8 l9 ~6 d4 ?7 u </object> ) }% m1 c! @3 Y/ W0 [4 w <object> & @" D& T: g8 g <name>apple</name> 6 |, R/ p6 l) w) ] <pose>Unspecified</pose> 8 B+ K. O/ Q* M& Z3 h+ d <truncated>0</truncated>2 z3 w* j6 Y) u. n5 F/ I
<difficult>0</difficult>/ y$ c3 V3 ]" X( ?# Z
<bndbox> , n7 R; ^" b* J0 @' L* B4 n <xmin>468</xmin>4 Y! U& N, s$ E( m |: r4 T# ^
<ymin>179</ymin> ) D( }& E1 z' X+ b <xmax>727</xmax>7 ? g0 G$ U/ o8 ^ U
<ymax>467</ymax>! G: }7 g. B6 k8 N- @
</bndbox> 0 q4 H! ]6 n, C" {) x </object> " W0 B8 P5 S6 W' V4 N <object> 7 X" D% M: N1 a; m& V <name>apple</name>( X7 R* M* h; W K% l: N0 Z
<pose>Unspecified</pose>8 G* T6 W: @7 E8 C/ l; s4 }
<truncated>1</truncated>8 G$ C' w5 p( |. J( o7 u, s$ B
<difficult>0</difficult> ( c; G! I* u/ I0 \ <bndbox>9 L, ?" U3 X0 _& y6 V
<xmin>1</xmin>* P# g2 S4 @6 w: j8 g; t
<ymin>63</ymin>' V9 g: n" X. J" W! D2 |
<xmax>308</xmax>2 i$ l2 M$ Q0 f$ C- Y) _ G
<ymax>414</ymax> - p: O) q' }8 U7 q* h+ A: S" U </bndbox>* c6 v" J$ U* V+ p
</object> 0 k$ J2 @( m6 l; l/ O8 G) }</annotation> ( m8 i& ~% |/ l; b1 * O. n3 [" p8 |' j2 6 l9 u) ]0 u) c0 S1 e) A; w w& p* C3) D4 L- v' b3 \) K8 x5 y
4 : `0 r: o% `- i& g$ r* R) Y I5 7 O+ w* ?& K9 k* o& Y69 N4 t: }( u1 x5 Q1 t) m
7 0 @; r. O: A% F6 l) e) {$ T8- y$ r6 x& [* R2 v! {6 j( T
9 - J) E( S, M- g9 a+ I10 . \6 J# M3 @0 A11 2 D6 M% E. ?3 r V& c12. T, p% ?# j8 w0 V
13 * A: A) b* c3 _: W5 V- k- A. {14$ j$ D/ h9 M/ h7 h* Z2 L* J
15' N7 B( r# b9 j! K: A; b
16) L* Z8 \7 X' X5 v' r0 ]
17 E1 d8 e( X+ H/ ?! h+ |. E18 ) o6 ? K ^7 ^* U' {4 Y* a% Z! R19. ?$ ?% a( d! |1 d* B
20( M: X9 `% ?) ~5 [2 a
21. U3 A4 y1 Y+ ^( m" u/ m
22# t6 k- ], @" @( m: T
233 H# o9 v i) X: R' \* T
24) {2 x/ C' B7 L& x8 O( {7 s! ?
258 c& Y. u+ J0 `$ q4 u5 R
26/ y- {* X1 n5 [! ~) F9 S( L+ U4 f
27 7 N+ E; } D( l9 B286 G) }1 ?0 p! E9 k3 e8 n
29 c U3 c6 B+ G; J9 s
30, |. E0 g( ^2 Z$ K- \* F
31 " L" c" j! G& c% q4 j, i328 \. T+ \; E5 M# s
33 ; M3 V; U. H4 V' {. `0 ~34# j. j; f$ E% g% {- |
35 5 @. F7 `0 W+ M2 w364 u1 W3 M4 t4 ?1 P3 u
37 . [1 o" E0 S* R) s. \# f38. B( Z4 ?+ v5 \6 n) b
39 ! Z; p7 L# F; t5 g- ~) D4 q; M40 3 g2 s8 l+ l! k" Q415 g, H+ |: S+ ^8 H& z3 A
426 y( Y, W B& Z( Z6 R4 S) f
433 V3 P- _% U" h: e
44 1 Q0 N4 J4 N+ a0 c/ j. N6 q45+ O, J* W& E! D! B! }: d
463 g7 c4 A1 Z% E
47) T2 ?) |: I. @# B
48: D+ W' X! ^6 `! S! w- i
49 4 s5 a9 M- S5 Z% d: p! B9 h50 4 W" f4 x$ }7 D% R t51 u6 h! r& |$ M X% s526 F3 H9 D' c; | Z1 f0 }8 q- C
539 B. n* E; {2 I$ A" S
54 6 H) P/ I" h/ b55 |$ X5 ]% k# }56 - o. x$ a( f1 W( @8 R57 4 u7 J2 A. L% k. b, G58 2 x; w8 V, y- Z. i D598 b+ F6 V+ w6 F! f/ V2 y7 X
60' C. |2 U* @! m; Z& `
61! p' A. y U! u0 M( @6 s" H9 M
62! D$ e/ {7 B. p+ y# f8 J3 _6 G/ B
63: K1 d" L/ I) G/ S
64+ v- c$ ^' d1 S J1 C3 T& A1 W
650 c& Z0 c: Z4 Z" Q; s
66 6 P& Q- D' L- |) i$ ~4 T4 k67* D3 M" o O! h8 d
683 p4 c5 e7 c% w, V# h5 s
696 Z, H9 |7 o' q1 Y
703 @4 B3 ?2 V3 T" s
71 & E7 y5 ]% t0 i2 Z. F$ ~: Q) e72 0 f2 F5 d$ U8 f# n, T' ~73 {3 K1 Y0 `- t: e! S% t747 @2 [: |$ j7 j0 D5 @
将xml文件提取图像信息,主要使用xml和opencv,基于torch提取,代码比较凌乱。 & n7 h5 i9 [! q/ R K % i& |+ p2 z3 `4 L 2 I$ l) D; b% ^/ I' simport os& t6 c! w1 E3 Z* Q
import numpy as np $ }2 J. ^+ `( T" _* Jimport cv2 ; Z' I( b3 M1 n8 j1 P; f: |( |2 Timport torch3 b+ f: Z$ z% t, U6 W; L
import matplotlib.patches as patches * m5 [$ Z: n0 _* a& k) Fimport albumentations as A' U8 _$ w" ~8 C9 H4 `7 B
from albumentations.pytorch.transforms import ToTensorV2' Z5 J7 S" h7 I/ C( u5 P
from matplotlib import pyplot as plt 4 `0 h8 ~5 R$ ~" b* Nfrom torch.utils.data import Dataset7 P: A/ D) B: ~
from xml.etree import ElementTree as et6 T( d- T0 F& \: `& Q5 n; E3 ?
from torchvision import transforms as torchtrans 7 r# G; a( C) a1 s9 _" Y; M, W& w* I; Q% r
" b) y3 _2 x; _# defining the files directory and testing directory6 S+ u8 s, N: ~
train_image_dir = 'train/train/image'4 l$ F& L( {: K1 H1 b
train_xml_dir = 'train/train/xml'+ S; G8 y# x4 g/ |# S) O
# test_image_dir = 'test/test/image' * q) B# k4 R& e. G$ l3 P) |2 ?# test_xml_dir = 'test/test/xml' 3 S5 N7 k6 I: }& ` , a9 g$ d9 p I! ]# I: y6 Z " X& B# ~/ y; l i3 X0 C1 @; G" `class FruitImagesDataset(Dataset):- b5 s7 ^. K- S
9 f' \; {/ n$ M* m) |
0 Z( B5 k* u! S8 Y
def __init__(self, image_dir, xml_dir, width, height, transforms=None):2 o4 r$ z% ^4 D
self.transforms = transforms1 A% B2 r& L2 D, Z8 N
self.image_dir = image_dir ! K k3 I2 c8 R1 I- x self.xml_dir = xml_dir: H" A- H' w, _- {
self.height = height% q H9 D l! B$ j5 ]5 @
self.width = width & d5 A N* {+ T% ? {. `% I6 s m" z9 r9 S4 c
/ ~! H) ?, G$ q6 t' T1 t # sorting the images for consistency ! s6 H+ _+ A' ?# Q( q3 K # To get images, the extension of the filename is checked to be jpg3 v+ S: \: S% H4 E) U
self.imgs = [image for image in os.listdir(self.image_dir) o; o. o/ q z if image[-4:] == '.jpg'] 8 `/ v2 Y4 O. W6 H+ ]* {% |2 P self.xmls = [xml for xml in os.listdir(self.xml_dir) - m, b+ Z$ t D, {: Y7 L) M l4 k! D9 W( c if xml[-4:] == '.xml']5 A% T0 g4 X: |% K
9 I& _, @) ]6 s: C" P a7 d- D
; q) |, u$ a. C: l- f # classes: 0 index is reserved for background6 q0 _% ~6 o1 j! ^! L
self.classes = ['apple', 'banana', 'orange'] 3 j! P, d* f1 u8 a ( z4 h) z9 @9 ]" R0 P0 g/ y9 M' t+ X9 Q) \$ c
def __getitem__(self, idx):" e5 U( G4 [( [
2 M& d4 |! J$ @% t. |' s$ c6 }. d k% P X! U
img_name = self.imgs[idx]4 t5 N) Z# z" P3 ?
image_path = os.path.join(self.image_dir, img_name), Y% ]# {2 n2 D6 b8 Z8 s) D' m
) w O- P r& H
3 U' C) M; I8 m
# reading the images and converting them to correct size and color & f U5 L: |: r4 a1 x img = cv2.imread(image_path)& A! r7 h0 w& w9 y8 ]: n2 E
img_rgb = cv2.cvtColor(img, cv2.COLOR_BGR2RGB).astype(np.float32)* _5 a z2 ~" d2 {% j
img_res = cv2.resize(img_rgb, (self.width, self.height), cv2.INTER_AREA). z# @/ @9 M/ v
# diving by 255' z9 x# B7 U- V1 F& L
img_res /= 255.0 4 f! G! I c! i. }! @5 m1 Q+ y. g' r: d
0 u# f0 e1 n6 V8 I! ^5 G4 p1 f# a
# annotation file % f: A7 u( I( S% k8 h annot_filename = img_name[:-4] + '.xml'7 Y r" A$ f" K* v- s( r
annot_file_path = os.path.join(self.xml_dir, annot_filename) # J7 R* a% y- a# [6 [$ @$ B! k& |% m6 q5 r% H0 T. ^: U; X
8 B+ Q; O/ B: o" ? boxes = []3 O! D$ E' _) c, D
labels = [] / E( \3 ~6 S- `. l5 S* f+ { tree = et.parse(annot_file_path) ! ^, D1 b) f+ t, Z% D: K% p5 j) s) x root = tree.getroot(); S& A2 a" l* `5 j; Z- h9 E& K/ e
* F9 Y ~ Y6 r* I1 ?! S& L% d
* S% I i! R5 Q- T. a # cv2 image gives size as height x width0 G/ }% Z9 C- g! ]7 T" S/ E& a
wt = img.shape[1] ' _* R2 p5 N: ^( A/ } ht = img.shape[0]( o% A* ?( |( V4 ?7 Z9 C
' V7 P& P2 N! r. J! i3 X1 t & r4 ~' R% \3 f8 Y8 T" n5 x! i( { # box coordinates for xml files are extracted and corrected for image size given$ J) U+ v# }) J. ?* W$ m
for member in root.findall('object'):& q8 K" e# f* y9 e9 h, r% y
labels.append(self.classes.index(member.find('name').text)) 9 }1 N: m' n. w0 e+ \ 4 a& _' m. I5 m4 X4 X. m& b: }( }. s" x4 o, ?
# bounding box s4 X% K2 H6 J$ F: y9 |
xmin = int(member.find('bndbox').find('xmin').text) 0 q1 D* z# s }# S* p/ e xmax = int(member.find('bndbox').find('xmax').text) 3 k4 T( @+ j+ k, d9 j& H! _. U ' s1 H5 v/ M ?* ~- k% M # S+ j4 @# ?4 {: ]+ ]' G1 L5 R ymin = int(member.find('bndbox').find('ymin').text); B; U h" F2 R4 J9 I4 {. z9 Q: n9 H
ymax = int(member.find('bndbox').find('ymax').text) " {$ ^" }! b o" J7 S" s* U- m. t- Y8 ^6 U- G( ]; R& A( y
4 _9 \+ c, `" o9 j; l- B
xmin_corr = (xmin / wt) * self.width 3 C3 O/ |. S O W! C8 L xmax_corr = (xmax / wt) * self.width% b3 U4 }, E, ]# n# D8 L6 J
ymin_corr = (ymin / ht) * self.height) A8 Z- {6 E) E$ G. N$ t
ymax_corr = (ymax / ht) * self.height % t+ T0 I) o+ c9 y* c boxes.append([xmin_corr, ymin_corr, xmax_corr, ymax_corr]) 1 g) z. y1 K! A8 X- b 0 A5 @+ g2 `) ^1 z% J$ r $ Q+ L+ U+ }1 y! ]7 c9 m+ L0 T/ l # convert boxes into a torch.Tensor. |1 ?% h" y/ B4 J/ }
boxes = torch.as_tensor(boxes, dtype=torch.float32)7 ?* D, t2 d" O9 p/ q- e% G
- O) j: e5 K9 \, M
- w2 u6 i- X8 Z/ g
# getting the areas of the boxes ! I) B, f. a4 ~ |9 d0 O area = (boxes[:, 3] - boxes[:, 1]) * (boxes[:, 2] - boxes[:, 0])3 G9 p/ L0 z* ^2 E
* m s! c/ X( D. _6 g, A* d! b/ s- X
# suppose all instances are not crowd/ U0 p/ j8 ?' B& b$ Y0 s, P
iscrowd = torch.zeros((boxes.shape[0],), dtype=torch.int64) 9 R. ~. \! v& P# S! K2 D# i* ]6 j. @' Z
C! F! `! d" g) `7 i, l& {. L: u/ Y
labels = torch.as_tensor(labels, dtype=torch.int64)- t X1 S2 k3 _) Y; U! M" d3 {/ g/ \; Y
] X/ x/ B; T) ]3 R- J% D; w
6 ^$ K% ^, D; l& ^; k" l, L0 I target = {} 5 L* ^; T9 b: p# ]) b/ f/ c$ D target["boxes"] = boxes , D7 }" j) t( X( B target["labels"] = labels ; i' @* d, d! | target["area"] = area& S; L# o! H- H% S3 ^! `; z
target["iscrowd"] = iscrowd % R# T) V/ ?/ x! o9 b # image_id " L5 j/ ^/ q; \3 S* @ image_id = torch.tensor([idx]) - Q$ C" ^4 q y: i target["image_id"] = image_id/ \3 O5 H8 u! q! {! {
3 ]* c% |6 k" I: Q0 {. p
T: T, b% b# c) \9 ~& X- V
if self.transforms:) Q. V0 p' Y1 g: v
sample = self.transforms(image=img_res,0 p( [' i% r4 D. G5 [
bboxes=target['boxes'], " u6 D; }) C- T1 J labels=labels)) B% y* a( E7 x" J- w2 W& E
2 f1 p/ ]+ U: G& V4 Z
r$ S( @" @/ Y img_res = sample['image'] # w: o9 h' ?' U) X8 s target['boxes'] = torch.Tensor(sample['bboxes'])" j& _- J4 Z) Y4 u
; T) n) k) c9 V, m( j! j' P S$ D- v, T7 g
return img_res, target/ t' R1 t0 F, D, Z |& h
/ E; D8 z4 b j2 A0 V
: H5 O8 d+ Q$ s! |- M" E7 y7 o4 b) f( Z def __len__(self):: y% ]. R/ z+ e/ @
return len(self.imgs) & t4 W" G% A& B: k# I# F7 h' j6 n t! y# Z
$ E2 `$ ]6 A# w1 c6 D7 ?- ^# function to convert a torchtensor back to PIL image 1 {9 P, k" b* q7 \3 U. Zdef torch_to_pil(img): 4 j+ Q/ }' a7 E1 _* v% C return torchtrans.ToPILImage()(img).convert('RGB') 8 C( E! k" O0 k1 i: o5 [5 _3 ]4 Q" q" J" u; v- Y
" x9 E( |2 L0 C3 R& D
& c/ m; |9 o1 a6 {: J/ V
4 w ^" x9 t6 \- }def plot_img_bbox(img, target): ; C' w2 w7 T2 A7 A # plot the image and bboxes+ v6 d6 u7 x0 R( C& m) S
fig, a = plt.subplots(1, 1) ) L4 }: V- m3 a/ Y4 E: C" L" e2 |9 d fig.set_size_inches(5, 5)+ X' K. S+ Y$ Q- w5 b1 @" S
a.imshow(img)' ?+ S- s; L, V# b
for box in (target['boxes']): . I4 p! a& V) A$ E# b( { x, y, width, height = box[0], box[1], box[2] - box[0], box[3] - box[1] * v) e) g9 M0 b* w rect = patches.Rectangle((x, y),$ d% ?) K# s8 J" S! B# X! X6 i
width, height, 6 e8 `" g: m: [5 k linewidth=2, g6 a# ]1 G3 w# T; f8 A+ g5 n: i
edgecolor='r', M; }& [1 E$ E
facecolor='none') 8 U& V) G1 X+ e( R" w% K 1 C+ h7 A* {- T, [, o3 O6 ?+ ]3 F8 r, \) q
# Draw the bounding box on top of the image2 Q. T6 m4 w# U6 D5 m3 U+ u9 O
a.add_patch(rect)( J5 ~- Z$ E% c2 ^+ h! Z
plt.show() 2 j0 ^3 |6 D( m4 r% v* y2 K6 G( S4 r5 w- T1 j+ {' @- M
, I, n% w8 ^: u! R; y0 E: ? 7 N& X" {3 O! f1 e2 z! d, e: k, i6 s# K. a: W ]% ~
def get_transform(train):* f* m+ @: |0 }
if train:' [$ i$ t- s! K6 x. g2 |' t( f& H' M
return A.Compose([$ q7 G8 \& W& j8 E% ]9 u
A.HorizontalFlip(0.5), # u5 W' ~3 @2 l2 q # ToTensorV2 converts image to pytorch tensor without div by 255 2 |: P7 D: P2 n; W9 e ToTensorV2(p=1.0) . |1 U6 l6 n/ }; [ ], bbox_params={'format': 'pascal_voc', 'label_fields': ['labels']}) ! E. ~5 `' { _3 k# w else:2 n' b8 }2 F0 F& T; x# A
return A.Compose([ 8 h4 @/ E5 x: o ToTensorV2(p=1.0) / [ Q$ p/ N, s4 m4 } ], bbox_params={'format': 'pascal_voc', 'label_fields': ['labels']})" y/ \$ k/ c b
) _4 l6 m; ^% X( h
/ O- k0 f, z$ J, c4 ?$ D
4 Q2 W1 {. u! c+ I% N2 r