SAEHD:

Maximum resolution is increased to 640. ‘hd’ archi is removed. ‘hd’ was experimental archi created to remove subpixel shake, but ‘lr_dropout’ and ‘disable random warping’ do that better. ‘uhd’ is renamed to ‘-u’ dfuhd and liaeuhd will be automatically renamed to df-u and liae-u in existing models. Added new experimental archi (key -d) which doubles the resolution using the same computation cost. It is mean same configs will be x2 faster, or for example you can set 448 resolution and it will train as 224. Strongly recommended not to train from scratch and use pretrained models. New archi naming: 'df' keeps more identity-preserved face. 'liae' can fix overly different face shapes. '-u' increased likeness of the face. '-d' (experimental) doubling the resolution using the same computation cost Examples: df, liae, df-d, df-ud, liae-ud, ... Improved GAN training (GAN_power option). It was used for dst model, but actually we don’t need it for dst. Instead, a second src GAN model with x2 smaller patch size was added, so the overall quality for hi-res models should be higher. Added option ‘Uniform yaw distribution of samples (y/n)’: Helps to fix blurry side faces due to small amount of them in the faceset. Quick96: Now based on df-ud archi and 20% faster. XSeg trainer: Improved sample generator. Now it randomly adds the background from other samples. Result is reduced chance of random mask noise on the area outside the face. Now you can specify ‘batch_size’ in range 2-16. Reduced size of samples with applied XSeg mask. Thus size of packed samples with applied xseg mask is also reduced.
2025-08-20 13:33:24 -07:00 · 2020-06-19 09:45:55 +04:00 · 2020-06-19 09:45:55 +04:00 · 0c2e1c3944
commit 0c2e1c3944
parent 9fd3a9ff8d
14 changed files with 513 additions and 572 deletions
--- a/samplelib/Sample.py
+++ b/samplelib/Sample.py
@ -5,8 +5,8 @@ import cv2
 import numpy as np

 from core.cv2ex import *
-from DFLIMG import *
 from facelib import LandmarksProcessor
+from core import imagelib
 from core.imagelib import SegIEPolys

 class SampleType(IntEnum):
@ -28,6 +28,7 @@ class Sample(object):
                 'landmarks',
                 'seg_ie_polys',
                 'xseg_mask',
+                 'xseg_mask_compressed',
                 'eyebrows_expand_mod',
                 'source_filename',
                 'person_name',
@ -42,6 +43,7 @@ class Sample(object):
                       landmarks=None,
                       seg_ie_polys=None,
                       xseg_mask=None,
+                       xseg_mask_compressed=None,
                       eyebrows_expand_mod=None,
                       source_filename=None,
                       person_name=None,
@ -60,6 +62,16 @@ class Sample(object):
            self.seg_ie_polys = SegIEPolys.load(seg_ie_polys)
        
        self.xseg_mask = xseg_mask
+        self.xseg_mask_compressed = xseg_mask_compressed
+        
+        if self.xseg_mask_compressed is None and self.xseg_mask is not None:
+            xseg_mask = np.clip( imagelib.normalize_channels(xseg_mask, 1)*255, 0, 255 ).astype(np.uint8)        
+            ret, xseg_mask_compressed = cv2.imencode('.png', xseg_mask)
+            if not ret:
+                raise Exception("Sample(): unable to generate xseg_mask_compressed")
+            self.xseg_mask_compressed = xseg_mask_compressed
+            self.xseg_mask = None
+ 
        self.eyebrows_expand_mod = eyebrows_expand_mod if eyebrows_expand_mod is not None else 1.0
        self.source_filename = source_filename
        self.person_name = person_name
@ -67,6 +79,14 @@ class Sample(object):

        self._filename_offset_size = None

+    def get_xseg_mask(self):
+        if self.xseg_mask_compressed is not None:
+            xseg_mask = cv2.imdecode(self.xseg_mask_compressed, cv2.IMREAD_UNCHANGED)
+            if len(xseg_mask.shape) == 2:
+                xseg_mask = xseg_mask[...,None]
+            return xseg_mask.astype(np.float32) / 255.0
+        return self.xseg_mask
+        
    def get_pitch_yaw_roll(self):
        if self.pitch_yaw_roll is None:
            self.pitch_yaw_roll = LandmarksProcessor.estimate_pitch_yaw_roll(self.landmarks, size=self.shape[1])
@ -97,6 +117,7 @@ class Sample(object):
                'landmarks': self.landmarks.tolist(),
                'seg_ie_polys': self.seg_ie_polys.dump(),
                'xseg_mask' : self.xseg_mask,
+                'xseg_mask_compressed' : self.xseg_mask_compressed,
                'eyebrows_expand_mod': self.eyebrows_expand_mod,
                'source_filename': self.source_filename,
                'person_name': self.person_name
--- a/samplelib/SampleGeneratorFace.py
+++ b/samplelib/SampleGeneratorFace.py
@ -6,11 +6,13 @@ import cv2
 import numpy as np

 from core import mplib
+from core.interact import interact as io
 from core.joblib import SubprocessGenerator, ThisThreadGenerator
 from facelib import LandmarksProcessor
 from samplelib import (SampleGeneratorBase, SampleLoader, SampleProcessor,
                       SampleType)

+
 '''
 arg
 output_sample_types = [
@ -23,15 +25,15 @@ class SampleGeneratorFace(SampleGeneratorBase):
                        random_ct_samples_path=None,
                        sample_process_options=SampleProcessor.Options(),
                        output_sample_types=[],
-                        add_sample_idx=False,
+                        uniform_yaw_distribution=False,
                        generators_count=4,
-                        raise_on_no_data=True,
+                        raise_on_no_data=True,                        
                        **kwargs):

        super().__init__(debug, batch_size)
+        self.initialized = False
        self.sample_process_options = sample_process_options
        self.output_sample_types = output_sample_types
-        self.add_sample_idx = add_sample_idx
        
        if self.debug:
            self.generators_count = 1
@ -40,15 +42,40 @@ class SampleGeneratorFace(SampleGeneratorBase):

        samples = SampleLoader.load (SampleType.FACE, samples_path)
        self.samples_len = len(samples)
-
-        self.initialized = False
+        
        if self.samples_len == 0:
            if raise_on_no_data:
                raise ValueError('No training data provided.')
            else:
                return
+                
+        if uniform_yaw_distribution:
+            samples_pyr = [ ( idx, sample.get_pitch_yaw_roll() ) for idx, sample in enumerate(samples) ]
+            
+            grads = 128
+            #instead of math.pi / 2, using -1.2,+1.2 because actually maximum yaw for 2DFAN landmarks are -1.2+1.2
+            grads_space = np.linspace (-1.2, 1.2,grads)

-        index_host = mplib.IndexHost(self.samples_len)
+            yaws_sample_list = [None]*grads
+            for g in io.progress_bar_generator ( range(grads), "Sort by yaw"):
+                yaw = grads_space[g]
+                next_yaw = grads_space[g+1] if g < grads-1 else yaw
+
+                yaw_samples = []
+                for idx, pyr in samples_pyr:
+                    s_yaw = -pyr[1]
+                    if (g == 0          and s_yaw < next_yaw) or \
+                    (g < grads-1     and s_yaw >= yaw and s_yaw < next_yaw) or \
+                    (g == grads-1    and s_yaw >= yaw):
+                        yaw_samples += [ idx ]
+                if len(yaw_samples) > 0:
+                    yaws_sample_list[g] = yaw_samples
+            
+            yaws_sample_list = [ y for y in yaws_sample_list if y is not None ]
+            
+            index_host = mplib.Index2DHost( yaws_sample_list )
+        else:
+            index_host = mplib.IndexHost(self.samples_len)

        if random_ct_samples_path is not None:
            ct_samples = SampleLoader.load (SampleType.FACE, random_ct_samples_path)
@ -110,14 +137,8 @@ class SampleGeneratorFace(SampleGeneratorBase):

                if batches is None:
                    batches = [ [] for _ in range(len(x)) ]
-                    if self.add_sample_idx:
-                        batches += [ [] ]
-                        i_sample_idx = len(batches)-1

                for i in range(len(x)):
                    batches[i].append ( x[i] )

-                if self.add_sample_idx:
-                    batches[i_sample_idx].append (sample_idx)
-
            yield [ np.array(batch) for batch in batches]
--- a/samplelib/SampleGeneratorFacePerson.py
+++ b/samplelib/SampleGeneratorFacePerson.py
@ -12,6 +12,98 @@ from samplelib import (SampleGeneratorBase, SampleLoader, SampleProcessor,
                       SampleType)


+
+class Index2DHost():
+    """
+    Provides random shuffled 2D indexes for multiprocesses
+    """
+    def __init__(self, indexes2D):
+        self.sq = multiprocessing.Queue()
+        self.cqs = []
+        self.clis = []
+        self.thread = threading.Thread(target=self.host_thread, args=(indexes2D,) )
+        self.thread.daemon = True
+        self.thread.start()
+
+    def host_thread(self, indexes2D):
+        indexes_counts_len = len(indexes2D)
+
+        idxs = [*range(indexes_counts_len)]
+        idxs_2D = [None]*indexes_counts_len
+        shuffle_idxs = []
+        shuffle_idxs_2D = [None]*indexes_counts_len
+        for i in range(indexes_counts_len):
+            idxs_2D[i] = indexes2D[i]
+            shuffle_idxs_2D[i] = []
+
+        sq = self.sq
+
+        while True:
+            while not sq.empty():
+                obj = sq.get()
+                cq_id, cmd = obj[0], obj[1]
+
+                if cmd == 0: #get_1D
+                    count = obj[2]
+
+                    result = []
+                    for i in range(count):
+                        if len(shuffle_idxs) == 0:
+                            shuffle_idxs = idxs.copy()
+                            np.random.shuffle(shuffle_idxs)
+                        result.append(shuffle_idxs.pop())
+                    self.cqs[cq_id].put (result)
+                elif cmd == 1: #get_2D
+                    targ_idxs,count = obj[2], obj[3]
+                    result = []
+
+                    for targ_idx in targ_idxs:
+                        sub_idxs = []
+                        for i in range(count):
+                            ar = shuffle_idxs_2D[targ_idx]
+                            if len(ar) == 0:
+                                ar = shuffle_idxs_2D[targ_idx] = idxs_2D[targ_idx].copy()
+                                np.random.shuffle(ar)
+                            sub_idxs.append(ar.pop())
+                        result.append (sub_idxs)
+                    self.cqs[cq_id].put (result)
+
+            time.sleep(0.001)
+
+    def create_cli(self):
+        cq = multiprocessing.Queue()
+        self.cqs.append ( cq )
+        cq_id = len(self.cqs)-1
+        return Index2DHost.Cli(self.sq, cq, cq_id)
+
+    # disable pickling
+    def __getstate__(self):
+        return dict()
+    def __setstate__(self, d):
+        self.__dict__.update(d)
+
+    class Cli():
+        def __init__(self, sq, cq, cq_id):
+            self.sq = sq
+            self.cq = cq
+            self.cq_id = cq_id
+
+        def get_1D(self, count):
+            self.sq.put ( (self.cq_id,0, count) )
+
+            while True:
+                if not self.cq.empty():
+                    return self.cq.get()
+                time.sleep(0.001)
+
+        def get_2D(self, idxs, count):
+            self.sq.put ( (self.cq_id,1,idxs,count) )
+
+            while True:
+                if not self.cq.empty():
+                    return self.cq.get()
+                time.sleep(0.001)
+                
 '''
 arg
 output_sample_types = [
@ -45,7 +137,7 @@ class SampleGeneratorFacePerson(SampleGeneratorBase):
        for i,sample in enumerate(samples):
            persons_name_idxs[sample.person_name].append (i)
        indexes2D = [ persons_name_idxs[person_name] for person_name in unique_person_names ]
-        index2d_host = mplib.Index2DHost(indexes2D)
+        index2d_host = Index2DHost(indexes2D)

        if self.debug:
            self.generators_count = 1
--- a/samplelib/SampleGeneratorFaceXSeg.py
+++ b/samplelib/SampleGeneratorFaceXSeg.py
@ -25,7 +25,7 @@ class SampleGeneratorFaceXSeg(SampleGeneratorBase):

        samples = sum([ SampleLoader.load (SampleType.FACE, path) for path in paths ]  )
        seg_sample_idxs = SegmentedSampleFilterSubprocessor(samples).run()
-        
+
        seg_samples_len = len(seg_sample_idxs)
        if seg_samples_len == 0:
            raise Exception(f"No segmented faces found.")
@ -63,9 +63,10 @@ class SampleGeneratorFaceXSeg(SampleGeneratorBase):

    def batch_func(self, param ):
        samples, seg_sample_idxs, resolution, face_type, data_format = param
-   
+
        shuffle_idxs = []
-        
+        bg_shuffle_idxs = []
+
        random_flip = True
        rotation_range=[-10,10]
        scale_range=[-0.05, 0.05]
@ -76,6 +77,25 @@ class SampleGeneratorFaceXSeg(SampleGeneratorBase):
        motion_blur_chance, motion_blur_mb_max_size = 25, 5
        gaussian_blur_chance, gaussian_blur_kernel_max_size = 25, 5

+        def gen_img_mask(sample):
+            img = sample.load_bgr()
+            h,w,c = img.shape
+            mask = np.zeros ((h,w,1), dtype=np.float32)
+            sample.seg_ie_polys.overlay_mask(mask)
+
+            if face_type == sample.face_type:
+                if w != resolution:
+                    img = cv2.resize( img, (resolution, resolution), cv2.INTER_LANCZOS4 )
+                    mask = cv2.resize( mask, (resolution, resolution), cv2.INTER_LANCZOS4 )
+            else:
+                mat = LandmarksProcessor.get_transform_mat (sample.landmarks, resolution, face_type)
+                img  = cv2.warpAffine( img,  mat, (resolution,resolution), borderMode=cv2.BORDER_CONSTANT, flags=cv2.INTER_LANCZOS4 )
+                mask = cv2.warpAffine( mask, mat, (resolution,resolution), borderMode=cv2.BORDER_CONSTANT, flags=cv2.INTER_LANCZOS4 )
+
+            if len(mask.shape) == 2:
+                mask = mask[...,None]
+            return img, mask
+
        bs = self.batch_size
        while True:
            batches = [ [], [] ]
@ -86,30 +106,26 @@ class SampleGeneratorFaceXSeg(SampleGeneratorBase):
                    if len(shuffle_idxs) == 0:
                        shuffle_idxs = seg_sample_idxs.copy()
                        np.random.shuffle(shuffle_idxs)
-                    idx = shuffle_idxs.pop()
-                    
-                    sample = samples[idx] 
+                    sample = samples[shuffle_idxs.pop()]
+                    img, mask = gen_img_mask(sample)

-                    img = sample.load_bgr()
-                    h,w,c = img.shape
+                    if np.random.randint(2) == 0:

-                    mask = np.zeros ((h,w,1), dtype=np.float32)
-                    sample.seg_ie_polys.overlay_mask(mask)
+                        if len(bg_shuffle_idxs) == 0:
+                            bg_shuffle_idxs = seg_sample_idxs.copy()
+                            np.random.shuffle(bg_shuffle_idxs)
+                        bg_sample = samples[bg_shuffle_idxs.pop()]
+
+                        bg_img, bg_mask = gen_img_mask(bg_sample)
+
+                        bg_wp   = imagelib.gen_warp_params(resolution, True, rotation_range=[-180,180], scale_range=[-0.10, 0.10], tx_range=[-0.10, 0.10], ty_range=[-0.10, 0.10] )
+                        bg_img  = imagelib.warp_by_params (bg_wp, bg_img,  can_warp=False, can_transform=True, can_flip=True, border_replicate=False)
+                        bg_mask = imagelib.warp_by_params (bg_wp, bg_mask, can_warp=False, can_transform=True, can_flip=True, border_replicate=False)
+
+                        c_mask = (1-bg_mask) * (1-mask)
+                        img = img*(1-c_mask) + bg_img * c_mask

                    warp_params = imagelib.gen_warp_params(resolution, random_flip, rotation_range=rotation_range, scale_range=scale_range, tx_range=tx_range, ty_range=ty_range )
-
-                    if face_type == sample.face_type:
-                        if w != resolution:
-                            img = cv2.resize( img, (resolution, resolution), cv2.INTER_LANCZOS4 )
-                            mask = cv2.resize( mask, (resolution, resolution), cv2.INTER_LANCZOS4 )
-                    else:
-                        mat = LandmarksProcessor.get_transform_mat (sample.landmarks, resolution, face_type)
-                        img  = cv2.warpAffine( img,  mat, (resolution,resolution), borderMode=cv2.BORDER_CONSTANT, flags=cv2.INTER_LANCZOS4 )
-                        mask = cv2.warpAffine( mask, mat, (resolution,resolution), borderMode=cv2.BORDER_CONSTANT, flags=cv2.INTER_LANCZOS4 )
-
-                    if len(mask.shape) == 2:
-                        mask = mask[...,None]
-
                    img   = imagelib.warp_by_params (warp_params, img,  can_warp=True, can_transform=True, can_flip=True, border_replicate=False)
                    mask  = imagelib.warp_by_params (warp_params, mask, can_warp=True, can_transform=True, can_flip=True, border_replicate=False)

@ -120,13 +136,13 @@ class SampleGeneratorFaceXSeg(SampleGeneratorBase):

                    if np.random.randint(2) == 0:
                        img = imagelib.apply_random_hsv_shift(img, mask=sd.random_circle_faded ([resolution,resolution]))
-                    else:                    
+                    else:
                        img = imagelib.apply_random_rgb_levels(img, mask=sd.random_circle_faded ([resolution,resolution]))
-                    
+
                    img = imagelib.apply_random_motion_blur( img, motion_blur_chance, motion_blur_mb_max_size, mask=sd.random_circle_faded ([resolution,resolution]))
                    img = imagelib.apply_random_gaussian_blur( img, gaussian_blur_chance, gaussian_blur_kernel_max_size, mask=sd.random_circle_faded ([resolution,resolution]))
                    img = imagelib.apply_random_bilinear_resize( img, random_bilinear_resize_chance, random_bilinear_resize_max_size_per, mask=sd.random_circle_faded ([resolution,resolution]))
-                    
+
                    if data_format == "NCHW":
                        img = np.transpose(img, (2,0,1) )
                        mask = np.transpose(mask, (2,0,1) )
@ -140,12 +156,10 @@ class SampleGeneratorFaceXSeg(SampleGeneratorBase):

            yield [ np.array(batch) for batch in batches]

-
-
 class SegmentedSampleFilterSubprocessor(Subprocessor):
    #override
    def __init__(self, samples ):
-        self.samples = samples        
+        self.samples = samples
        self.samples_len = len(self.samples)

        self.idxs = [*range(self.samples_len)]
@ -164,7 +178,7 @@ class SegmentedSampleFilterSubprocessor(Subprocessor):
    #override
    def on_clients_finalized(self):
        io.progress_bar_close()
-        
+
    #override
    def get_data(self, host_dict):
        if len (self.idxs) > 0:
@ -184,11 +198,11 @@ class SegmentedSampleFilterSubprocessor(Subprocessor):
        io.progress_bar_inc(1)
    def get_result(self):
        return self.result
-        
+
    class Cli(Subprocessor.Cli):
        #overridable optional
        def on_initialize(self, client_dict):
            self.samples = client_dict['samples']
-            
+
        def process_data(self, idx):
            return idx, self.samples[idx].seg_ie_polys.get_pts_count() != 0
--- a/samplelib/SampleProcessor.py
+++ b/samplelib/SampleProcessor.py
@ -56,15 +56,16 @@ class SampleProcessor(object):
            ct_sample_bgr = None
            h,w,c = sample_bgr.shape
            
-            def get_full_face_mask():                                        
-                if sample.xseg_mask is not None:                   
-                    full_face_mask = sample.xseg_mask
-                    if full_face_mask.shape[0] != h or full_face_mask.shape[1] != w:
-                        full_face_mask = cv2.resize(full_face_mask, (w,h), interpolation=cv2.INTER_CUBIC)                    
-                        full_face_mask = imagelib.normalize_channels(full_face_mask, 1)
+            def get_full_face_mask():   
+                xseg_mask = sample.get_xseg_mask()                                     
+                if xseg_mask is not None:           
+                    if xseg_mask.shape[0] != h or xseg_mask.shape[1] != w:
+                        xseg_mask = cv2.resize(xseg_mask, (w,h), interpolation=cv2.INTER_CUBIC)                    
+                        xseg_mask = imagelib.normalize_channels(xseg_mask, 1)
+                    return np.clip(xseg_mask, 0, 1)
                else:
                    full_face_mask = LandmarksProcessor.get_image_hull_mask (sample_bgr.shape, sample_landmarks, eyebrows_expand_mod=sample.eyebrows_expand_mod )
-                return np.clip(full_face_mask, 0, 1)
+                    return np.clip(full_face_mask, 0, 1)
                
            def get_eyes_mask():
                eyes_mask = LandmarksProcessor.get_image_eye_mask (sample_bgr.shape, sample_landmarks)