It seems this prompt and the seaplane prompt have appeared with higher frequency recently; they must have been deployed to the production environment.

So naturally, I had to tackle it. The dataset can be found here .

Overview of CAPTCHA samples for the leaf elephant task

First, let’s analyze it. There are only three types in total: one made of leaves, one made of petals, and a third type made of some unknown metaphysical stuff that looks pitch black. Looking at the content composition, besides elephants, there are horses—it’s essentially a binary classification.

Actually, besides the prompt ["Please select all the elephants drawn with lеaves"], there is another similar one "Please select all the horses drawn with flowers"1, but I have almost never seen this prompt. I don’t know how the person who raised this issue managed to generate it. I think the bigger reason is that the perplexity of the flower images is too high, making them extremely hard to distinguish, which would greatly degrade the user experience. However, compared to humans extracting features, I feel machines can extract features much faster –.

Sample images of animals composed of flowers
Another set of sample images of animals composed of flowers
Blurry sample one among generated animal images
Blurry sample two among generated animal images

The approach is clear: first, the composition classification is as simple as it gets; we just need to extract the dominant color of the image. Here, I used kmeans for color clustering to extract the dominant color. I selected clustering centers k=3 to distinguish between bright and dark areas, so I assigned one center to each, leaving the third for the dominant color. This way, I only need to set a threshold to calculate the distance between the color and green. I chose 200, achieving a 100% discrimination accuracy.

 1
 2
 3
 4
 5
 6
 7
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
def _style_classification(self, img):
    # opencv to numpy
    img = np.array(img)
    # from 3d -> 2d
    img = img.reshape((img.shape[0] * img.shape[1], img.shape[2])).astype(np.float64)
    # print(img.shape)
    centroid, label = kmeans2(img, k=3)
    # print(centroid)
    # print(label)

    green_centroid = np.array([0.0, 255.0, 0.0])

    flag = False
    min_dis = np.inf
    for i in range(len(centroid)):
        # distance between centroid and green < threshold
        # print(np.linalg.norm(centroid[i] - green_centroid))
        min_dis = min(min_dis, np.linalg.norm(centroid[i] - green_centroid))

    if min_dis < 200:
        flag = True

    return flag

Distinguishing between elephants and horses is truly challenging, or maybe not challenging at all. Your image processing methods, such as further clustering or, like in the previous two blog posts [1]2[2]3, calculating weight or superpixel count, are quite difficult for this distinction. Since elephants and horses have similar volumes, and the ones below all have 5-6 support points (because elephants have trunks and tails, while horses have tails and mouths), it’s hard to tell them apart. To say it simply, isn’t this just "Dog vs Cat"? It’s merely an introductory image classification task in deep learning…

First, let’s label the data. We need to use that style of classifier to filter first, which will save us from labeling a lot of data.

 1
 2
 3
 4
 5
 6
 7
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
import os
import sys
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
import cv2
from src.services.hcaptcha_challenger.solutions.elephant_solution import ElephantSolution

img_path = os.path.join('elephants_drawn_with_leaves')
label_file = open('label.txt', 'w')
os.makedirs(os.path.join(img_path, 'elephant'), exist_ok=True)
os.makedirs(os.path.join(img_path, 'house'), exist_ok=True)

if __name__ == '__main__':
    # 0 for house, 1 for elephant
    imgs = os.listdir(img_path)
    edwls = ElephantSolution()
    for idx, img_ in enumerate(imgs):
        img_path_ = os.path.join(img_path, img_)
        if os.path.isdir(img_path_):
            continue
        img = cv2.imread(img_path_)
        cv2.imshow("img", img)

        print(f'{img_path_}: {idx}')

        if edwls._style_classification(img):
            key = cv2.waitKey(0)
            if key == ord('0'):
                label_file.write(f'{img_path_} 0\n')
                label_file.flush()
                cv2.imwrite(os.path.join(img_path, 'house', img_), img)
                print(f'{img_path_} 0: house')
            elif key == ord('1'):
                label_file.write(f'{img_path_} 1\n')
                label_file.flush()
                cv2.imwrite(os.path.join(img_path, 'elephant', img_), img)
                print(f'{img_path_} 1: elephant')
        else:
            print('Drop')

Design a simple ResNet model. There’s no need for a massive model like ResNet18; this problem doesn’t even warrant it, so I DIYed a very small model, and I even downscaled the images to (64 x 64), which significantly reduces the number of parameters.

Training and testing in one go.

  1
  2
  3
  4
  5
  6
  7
  8
  9
 10
 11
 12
 13
 14
 15
 16
 17
 18
 19
 20
 21
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
import os
import sys
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
import shutil
import torch
import torch.nn as nn
import torch.nn.functional as F
import torchvision

import cv2
from PIL import Image
from src.services.hcaptcha_challenger.solutions.elephant_solution import ElephantSolution


class ResidualBlock(nn.Module):

    def __init__(self, in_channels, out_channels, stride=1):
        super(ResidualBlock, self).__init__()
        self.conv1 = nn.Conv2d(in_channels,
                               out_channels,
                               kernel_size=3,
                               stride=stride,
                               padding=1,
                               bias=False)
        self.bn1 = nn.BatchNorm2d(out_channels)
        self.conv2 = nn.Conv2d(out_channels,
                               out_channels,
                               kernel_size=3,
                               stride=1,
                               padding=1,
                               bias=False)
        self.bn2 = nn.BatchNorm2d(out_channels)
        self.relu = nn.ReLU(inplace=True)

        self.downsample = nn.Sequential()
        if stride != 1 or in_channels != out_channels:
            self.downsample = nn.Sequential(
                nn.Conv2d(in_channels, out_channels, kernel_size=1, stride=stride, bias=False),
                nn.BatchNorm2d(out_channels))

    def forward(self, x):
        residual = x
        out = self.relu(self.bn1(self.conv1(x)))
        out = self.bn2(self.conv2(out))
        out += self.downsample(residual)
        out = self.relu(out)
        return out


class Net(nn.Module):

    def __init__(self, in_channels=3, num_classes=10):
        super(Net, self).__init__()
        self.conv1 = nn.Conv2d(in_channels, 16, kernel_size=7, stride=2, padding=3, bias=False)
        self.bn1 = nn.BatchNorm2d(16)
        self.relu = nn.ReLU(inplace=True)
        self.maxpool = nn.MaxPool2d(kernel_size=3, stride=2, padding=1)
        self.resblock1 = ResidualBlock(16, 32)
        self.resblock2 = ResidualBlock(32, 64, stride=2)
        self.avgpool = nn.AvgPool2d(kernel_size=7, stride=1)
        self.fc = nn.Linear(256, num_classes)

    def forward(self, x):
        x = self.conv1(x)
        x = self.bn1(x)
        x = self.relu(x)
        x = self.maxpool(x)
        x = self.resblock1(x)
        x = self.resblock2(x)
        x = self.avgpool(x)
        x = x.view(x.size(0), -1)
        # print(x.size())
        x = self.fc(x)
        return x


img_path = os.path.join('..', 'database', 'elephants_drawn_with_leaves')

img_transform = torchvision.transforms.Compose([
    # torchvision.transforms.Grayscale(num_output_channels=1),
    # torchvision.transforms.GaussianBlur(kernel_size=3),
    torchvision.transforms.Resize((64, 64)),
    torchvision.transforms.ToTensor(),
])


def train():
    model = Net(3, 2)
    model.train()
    model.cuda()
    optimizer = torch.optim.Adam(model.parameters(), lr=0.005)
    # focal loss
    criterion = nn.CrossEntropyLoss()

    print('model:', model)

    data = torchvision.datasets.ImageFolder(img_path, transform=img_transform)
    data_loader = torch.utils.data.DataLoader(data, batch_size=1, shuffle=True)
    print(f'{len(data)} images')
    epochs = 20

    # train with focal loss
    for epoch in range(epochs):
        total_loss = 0
        total_acc = 0
        for i, (img, label) in enumerate(data_loader):
            img = img.cuda()
            label = label.cuda()
            optimizer.zero_grad()
            out = model(img)
            loss = criterion(out, label)
            loss.backward()
            optimizer.step()
            if (i + 1) % 10 == 0:
                print(f'epoch: {epoch + 1}, iter: {i + 1}, loss: {loss.item():.4f}')
            total_loss += loss.item()
            total_acc += torch.sum(torch.argmax(out, dim=1) == label).item()
        print(
            f'epoch: {epoch + 1}, avg loss: {total_loss / len(data):.4f}, avg acc: {total_acc / len(data):.4f}'
        )

    torch.save(model.state_dict(), 'model.pth')


def test_single(model, img):

    img = img_transform(img)
    img = img.unsqueeze(0)
    img = img.cuda()
    out = model(img)
    pred = torch.argmax(out, dim=1)
    # print(f'pred: {pred.item()}')
    if pred.item() == 0:
        return 0
    else:
        return 1


def test():
    model = Net(3, 2)
    model.load_state_dict(torch.load('model.pth'))
    model.eval()
    torch.onnx.export(model,
                      torch.randn(1, 3, 64, 64),
                      'model.onnx',
                      verbose=True,
                      export_params=True)
    model.cuda()
    test_data_path = os.path.join('val-dataset')
    imgs = os.listdir(test_data_path)

    dir1 = os.path.join('val-dataset', 'elephant_drawn_with_leaves')
    dir2 = os.path.join('val-dataset', 'house_drawn_with_leaves')
    dir3 = os.path.join('val-dataset', 'without_leaves')

    dirs = [dir1, dir2, dir3]

    for dir in dirs:
        if os.path.exists(dir):
            shutil.rmtree(dir)
        os.mkdir(dir)
    es = ElephantSolution()

    for img in imgs:
        if os.path.isdir(os.path.join(test_data_path, img)):
            continue
        img_ = cv2.imread(os.path.join(test_data_path, img))
        result = 2
        if es._style_classification(img_):
            result = test_single(model, Image.open(os.path.join(test_data_path, img)))

        print(f'{img} is {result} save to {os.path.join(dirs[result], img)}')
        cv2.imwrite(os.path.join(dirs[result], img), img_)


if __name__ == '__main__':
    train()
    test()

Here is an interesting part: initially, after training, I tested it and found the error rate was quite high, around 10%, which is almost unacceptable because there are only 9 CAPTCHA images in total, making it tricky. I thought it was overfitting (this should be the conventional approach, after all, the training set accuracy was nearly 100%), so I increased the learning rate and decreased the number of epochs, but no matter what I did, the error rate stayed around 5%. I was baffled; how could such a simple classification task perform so poorly? Later… I surprisingly increased the number of epochs, and found that the test set accuracy also reached nearly 100%. Finally, the overall error rate on the test set was around 1.8%, so I didn’t continue optimizing, and I was even too lazy to put the test data back into training.

View original GIF

Finally, the entire model parameters were saved as pt, which is only 311KB. If exported as onnx, it becomes 290KB.

Here I learned another trick: after exporting as onnx, I can directly use opencv for inference, which can greatly save resources in the deployment environment.

Solution: The complete code is below

 1
 2
 3
 4
 5
 6
 7
 8
 9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
import os
import sys
sys.path.append(os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
import cv2
import numpy as np
from scipy.cluster.vq import kmeans2


class ElephantSolution:

    def __init__(self):
        self.debug = True

    def solution(self, img_stream, **kwargs) -> bool:  # noqa
        """Implementation process of solution"""
        img_arr = np.frombuffer(img_stream, np.uint8)
        img = cv2.imdecode(img_arr, flags=1)

        cv2.imshow("img", img)
        cv2.waitKey(0)

        if not self._style_classification(img):
            return False

        # using model to predict
        # print(img.shape)
        # resize
        img = cv2.resize(img, (64, 64))
        # print(img.shape)
        model_path = os.path.join('..', '..', '..', 'model', 'elephant_model.onnx')
        model = cv2.dnn.readNetFromONNX(model_path)
        blob = cv2.dnn.blobFromImage(img, 1 / 255.0, (64, 64), (0, 0, 0), swapRB=True, crop=False)
        model.setInput(blob)
        out = model.forward()
        # print(out.shape)
        # print(out)
        label = np.argmax(out, axis=1)[0]
        # print(label)
        if label == 0:
            return True

        return False

    def _style_classification(self, img):
        # opencv to numpy
        img = np.array(img)
        # from 3d -> 2d
        img = img.reshape((img.shape[0] * img.shape[1], img.shape[2])).astype(np.float64)
        # print(img.shape)
        centroid, label = kmeans2(img, k=3)
        # print(centroid)
        # print(label)

        green_centroid = np.array([0.0, 255.0, 0.0])

        flag = False
        min_dis = np.inf
        for i in range(len(centroid)):
            # distance between centroid and green < threshold
            # print(np.linalg.norm(centroid[i] - green_centroid))
            min_dis = min(min_dis, np.linalg.norm(centroid[i] - green_centroid))

        if min_dis < 200:
            flag = True

        return flag