Repository navigation
Expand file tree
/
Copy pathwebImageCrawler.py
More file actions
75 lines (65 loc) · 2.22 KB
/
Copy pathwebImageCrawler.py
File metadata and controls
75 lines (65 loc) · 2.22 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
import dload
from bs4 import BeautifulSoup
from selenium import webdriver
import time
import random
import category
import os
class WebImageCrawler:
def __init__(self, platform, counter):
self.platform = platform
self.counter = counter
self.resultPath = "./imgs"
self.driver = None
self.baseUrl = {
"daum":"https://search.daum.net/search?w=img&nil_search=btn&DA=NTB&enc=utf8&q="
}
# crawl images step by step
def crawl_imgs(self):
self.check_driver()
self.make_result_dir()
categories = self.get_categories()
self.get_img(categories)
# check the chrome driver
def check_driver(self):
try:
self.driver = webdriver.Chrome('./chromedriver')
except:
print("check the chrome driver path")
# make the result dir
def make_result_dir(self):
resultPath = self.resultPath
if not os.path.exists(resultPath):
os.makedirs(resultPath)
# get the WebImage from url
def get_img(self, categories):
for item in categories:
url = self.baseUrl[self.platform] + item
self.driver.get(url)
# get the thumnail info
req = self.driver.page_source
soup = BeautifulSoup(req, 'html.parser')
thumbnails = soup.select("#imgList > div > a > img")
# download the imgs
counter = 0
for thumbnail in thumbnails:
src = thumbnail["src"]
fname = self.generate_random_name()
dload.save(src, self.resultPath + "/" +fname)
counter+=1
if counter == self.counter:
break
self.driver.quit()
return
def get_categories(self):
cat = category.imagenet_catecory.values()
catecories = []
for items in cat:
item = items.split(", ")
for i in item:
catecories.append(i)
return catecories
def generate_random_name(self, format = '.jpg'):
chr_list = [chr(alpha) for alpha in range(97, 123)]
fname = ''.join(random.sample(chr_list, 10)) + format
return fname