Repository navigation
Expand file tree
/
Copy pathdataload.py
More file actions
320 lines (275 loc) · 12.3 KB
/
Copy pathdataload.py
File metadata and controls
320 lines (275 loc) · 12.3 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
#! /usr/bin/env python3
# -*- coding: utf-8 -*-
#
# Copyright(C) 2017-2020 T.WKVER | </MATRIX>. All rights reserved.
# code by </MATRIX>@Neod Anderjon(LeaderN)
#
# dataload.py
# Original Author: Neod Anderjon(1054465075@qq.com/EnatsuManabu@gmail.com), 2018-3-10
#
# PixivCrawlerIII component
# T.WKVER crawler data handler loader for PixivCrawlerIII project
# List all constant data
import time, os, re
# project info
PROJECT_NAME = 'PixivCrawlerIII'
DEVELOPER = 'Neod Anderjon(LeaderN)'
LABORATORY = 'T.WKVER'
ORGANIZATION = '</MATRIX>'
VERSION = '3.3.3'
# operation result status code
PUB_E_OK = 0
PUB_E_FAIL = -1
PUB_E_RESPONSE_FAIL = -2
PUB_E_PARAM_FAIL = -3
PUB_E_REGEX_FAIL = -4
# run mode
MODE_INTERACTIVE = '1'
MODE_SERVER = '2'
# rtn or ira
SELECT_RTN = '1'
SELECT_IRA = '2'
SELECT_HELP = '3'
SELECT_EXIT = '4'
# rtn daily | weekly | monthly
RANK_DAILY = '1'
RANK_WEEKLY = '2'
RANK_MONTHLY = '3'
# ranking top page type
PAGE_ORDINARY = '1'
PAGE_R18 = '2'
PAGE_R18G = '3'
# sex word
SEX_NORMAL = '0'
SEX_MALE = '1'
SEX_FEMALE = '2'
NORMAL = "\033[0m"
HL_CR = lambda pcode: "\033[0;31;40m" + pcode + NORMAL # code red, use in logo
BR_CB = lambda pcode: "\033[7;31m" + pcode + NORMAL # background red, use in error or failed operate
HL_CY = lambda pcode: "\033[0;33;40m" + pcode + NORMAL # code yellow, use in ask question
BY_CB = lambda pcode: "\033[7;33;44m" + pcode + NORMAL # code blue and background yellow, use in important info
# logfile log real-time operation
base_time = time.time()
# set time color effect to yellow code and blue background
realtime_logword = lambda bt: "\033[7;34;43m[%02d:%02d:%02d]\033[0m " \
% (int((time.time() - bt) / 3600),
int((time.time() - bt) / 60),
(time.time() - bt) % 60)
# log with time message operations
LT_INPUT = lambda str_: input(realtime_logword(base_time) + str_) # input param method with time log
LT_PRINT = lambda str_: print(realtime_logword(base_time) + str_) # print string method with time log
LT_FLUSH = lambda str_, *args_, **kwargs_: print(('\r' + \
realtime_logword(base_time) + str_).format(*args_, **kwargs_), end="") # flush simple line method with time log
GIF_TYPE_LABEL = '2' # pixiv use number 2 express gif type
def nolog_raise_arguerr():
"""Argument(s) error info
:return: none
"""
LT_PRINT(BR_CB('argument(s) error'))
def crawler_logo():
"""Print crawler logo
:return: none
"""
LT_PRINT(HL_CR(LABORATORY + ' ' + ORGANIZATION + ' technology support | Code by ' + ORGANIZATION + '@' + DEVELOPER))
SYSTEM_MAX_THREADS = 400 # setting system can contain max sub-threads
DEFAULT_PURE_PROXYDNS = '8.8.8.8:53' # default pure dns by Google
def platform_setting():
"""Get OS platform to set folder format
:return: platform work directory
"""
work_dir = None
home_dir = os.environ['HOME'] # get system default setting home folder, for windows
get_login_user = os.getlogin() # get login user name to build user home directory, for linux
# linux
if os.name == 'posix':
if get_login_user != 'root':
work_dir = '/home/' + get_login_user + '/Pictures/Crawler/'
else:
# if your run crawler program in Android Pydroid 3
# change here work_dir to /sdcard/Pictures/Crawler/
work_dir = '/sdcard/Pictures/Crawler/'
# windows
elif os.name == 'nt':
work_dir = home_dir + '/PictureDatabase/Crawler/'
else:
pass
return work_dir
# for filesystem operation entity
g_dl_work_dir = platform_setting()
# real time clock
_rtc = time.localtime()
_ymd = '%d-%d-%d' % (_rtc[0], _rtc[1], _rtc[2])
# AES encrypto use secret key
AES_SECRET_KEY = 'secretkeyfrommat'.encode('utf-8') # 16 bytes secret key
# universal path
LOGIN_AES_INI_PATH = os.getcwd() + '/.aes_crypto_login.ini'
LOG_NAME = '/CrawlerWork[%s].log' % _ymd
HTML_NAME = '/CrawlerWork[%s].html' % _ymd
RANK_DIR = g_dl_work_dir + 'rankingtop_%s/' % _ymd
LOG_PATH = RANK_DIR + LOG_NAME
HTML_PATH = RANK_DIR + HTML_NAME
# selenium method use, you need to replace to your own chromedriver path
chrome_user_data_dir = 'C:\\Users\\neod-anderjon\\AppData\\Local\\Google\\Chrome\\User Data'
local_cache_cookie_path = os.getcwd() + '/.pixiv_cookies.json'
# login and request image https proxy
WWW_HOST_URL = "www.pixiv.net"
HTTPS_HOST_URL = 'https://www.pixiv.net/'
ACCOUNTS_URL = "accounts.pixiv.net"
LOGIN_POSTKEY_URL = 'https://accounts.pixiv.net/login?lang=zh&source=pc&view_type=page&ref=wwwtop_accounts_index'
LOGIN_REQUEST_API_URL = "https://accounts.pixiv.net/api/login?lang=zh"
_LOGIN_REQUEST_URL = "https://accounts.pixiv.net"
_LOGIN_REQUEST_REF_URL = "https://accounts.pixiv.net/login"
# request universal original image constant words
ORIGINAL_IMAGE_HEAD = 'https://i.pximg.net/img-original/img'
ORIGINAL_IMAGE_TAIL = lambda px: '_p%d.png' % px
# ranking top mode url and word
RANKING_URL = 'http://www.pixiv.net/ranking.php?mode='
R18_WORD = '_r18'
DAILY_WORD = 'daily'
WEEKLY_WORD = 'weekly'
MONTHLY_WORD = 'monthly'
MALE_WORD = 'male'
FEMALE_WORD = 'female'
MALE_R18_WORD = 'male_r18'
FEMALE_R18_WORD = 'female_r18'
R18G_WORD = 'r18g'
RANK_DAILY_URL = RANKING_URL + DAILY_WORD
RANK_DAILY_MALE_URL = RANKING_URL + MALE_WORD
RANK_DAILY_FEMALE_URL = RANKING_URL + FEMALE_WORD
RANK_WEEKLY_URL = RANKING_URL + WEEKLY_WORD
RANK_MONTHLY_URL = RANKING_URL + MONTHLY_WORD
RANK_DAILY_R18_URL = RANK_DAILY_URL + R18_WORD
RANK_DAILY_MALE_R18_URL = RANKING_URL + MALE_R18_WORD
RANK_DAILY_FEMALE_R18_URL = RANKING_URL + FEMALE_R18_WORD
RANK_WEEKLY_R18_URL = RANK_WEEKLY_URL + R18_WORD
RANK_R18G_URL = RANKING_URL + R18G_WORD
# base artwork mainpage get origin image url
BASEPAGE_URL = lambda aid: 'http://pixiv.net/artworks/%s' % aid
USERS_ARTWORKS_URL = lambda iid: 'http://pixiv.net/users/%s/artworks' % iid
AJAX_ALL_URL = lambda aid: 'http://www.pixiv.net/ajax/user/%s/profile/all' % aid
IDS_UNIT = lambda iid: 'ids%%5B%%5D=%s&' % iid # unsequence "ids[]="
ALLREPOINFO_URL = lambda aid, ids_sym, is_first_page: \
'http://www.pixiv.net/ajax/user/%s/profile/illusts?%swork_category=illustManga&is_first_page=%d' % (aid, ids_sym, is_first_page)
ONE_PAGE_COMMIT = 48
JUDGE_NOGIF_WORD = '_p0_master1200.jpg' # ignore gif
FROM_URL_GET_IMG_NAME = lambda url: url[57:-4] # from url get image name
# http status code
HTTP_REP_OK_CODE = 200
HTTP_REP_403_CODE = 403
HTTP_REP_404_CODE = 404
# login headers info dict
# here is an example of two different operating systems for headers
# but in fact, the crawler can pretend to be the headers of any operating system
_USERAGENT_LINUX = ("Mozilla/5.0 (X11; Linux x86_64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/56.0.2924.87 Safari/537.36")
_USERAGENT_WIN = ("Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/75.0.3770.142 Safari/537.36")
_HEADERS_ACCEPT = "application/json"
_HEADERS_ACCEPT2 = ("text/html,application/xhtml+xml,application/xml;q=0.9,"
"image/webp,image/apng,*/*;q=0.8")
_HEADERS_ACCEPT_ENCODING = "gzip, deflate, br"
_HEADERS_ACCEPT_ENCODING2 = "br" # request speed slowly, but no error
## _HEADERS_ACCEPT_LANGUAGE = "en-US,en;q=0.8,zh-TW;q=0.6,zh;q=0.4,zh-CN;q=0.2"
_HEADERS_ACCEPT_LANGUAGE = "zh-CN,zh;q=0.9"
_HEADERS_CONTENT_TYPE = "application/x-www-form-urlencoded"
_HEADERS_CONNECTION = 'keep-alive'
# some regex words depend on website url or webpage source
# if website update or change them, regex need to be updated
POSTKEY_REGEX = 'key".*?"(.*?)"'
# group match info
RANKING_INFO_REGEX = ('data-rank-text="(.*?)" data-title="(.*?)" data-user-name="(.*?)"'
'.*?data-id="(.*?)".*?data-user-id="(.*?)"')
NUMBER_REGEX = '\d+\.?\d*' # general number match
DATASRC_REGEX = 'data-src="(.*?)"'
ILLUST_NAME_REGEX = lambda iid: '"userId":"%s","name":"(.*?)","image"' % iid
AJAX_ALL_IDLIST_REGEX = '"(.*?)":null'
PAGE_REQUEST_SYM_REGEX = '"error":(.*?),'
PAGE_TGT_INFO_SQUARE_REGEX = '"id":"(.*?)","title":"(.*?)"(.*?)"url":"(.*?)1200.jpg"(.*?)"pageCount":(.*?),' # support square & custom label
ILLUST_TYPE_REGEX = '"illustType":(.*?),'
SPAN_REGEX = '<span>(.*?)</span>'
RANKING_SECTION_REGEX = '<section id=(.*?)</section>'
LOGIN_INFO_REGEX = 'error":(.*?),"message'
#### code by CSDN@orangleliu
EMOJI_REGEX = (u"(\ud83d[\ude00-\ude4f])|" # emoticons
u"(\ud83c[\udf00-\uffff])|" # symbols & pictographs (1 of 2)
u"(\ud83d[\u0000-\uddff])|" # symbols & pictographs (2 of 2)
u"(\ud83d[\ude80-\udeff])|" # transport & map symbols
u"(\ud83c[\udde0-\uddff])" # flags (iOS)
"+")
emoji_pattern = re.compile(EMOJI_REGEX, re.S)
EMOJI_REPLACE = lambda _str: emoji_pattern.sub('[EMOJI]', _str)
UNICODE_ESCAPE = lambda _raw_str: _raw_str.encode('utf-8').decode('unicode_escape')
def dict2list (input_dict):
"""Change dict data-type to list
:param input_dict: dict
:return: list
"""
result_list = []
for key, value in list(input_dict.items()):
item = (key, value)
result_list.append(item)
return result_list
def uc_user_agent():
"""Choose platform user-agent headers
In fact, which agent can be selected
It is recommended to directly select the Windows version
:return: headers
"""
# build dict word
ua_headers_linux = {'User-Agent': _USERAGENT_LINUX}
ua_headers_windows = {'User-Agent': _USERAGENT_WIN}
# platform choose
headers = None
if os.name == 'posix':
headers = ua_headers_linux
elif os.name == 'nt':
headers = ua_headers_windows
else:
pass
return headers
def build_login_headers(cookie):
"""Build the first request login headers
Actually this function has not be called
If login function in API(wkvcwapi.wca_camouflage_login) call this function,
then response will get a boolean False
:param cookie: cookie
:return: login headers
"""
base_headers = {
':authority': ACCOUNTS_URL,
':method': "POST",
':path': "/api/login?lang=zh",
':scheme': "https",
'accept': _HEADERS_ACCEPT,
'accept-encoding': _HEADERS_ACCEPT_ENCODING,
'accept-language': _HEADERS_ACCEPT_LANGUAGE,
'content-length': "546",
'content-type': _HEADERS_CONTENT_TYPE,
'cookie': cookie,
'dnt': "1",
'origin': _LOGIN_REQUEST_URL,
'referer': _LOGIN_REQUEST_REF_URL,
'sec-fetch-mode': "cors",
'sec-fetch-site': "same-origin"
}
# dict merge, longth-change argument
build_headers = dict(base_headers, **uc_user_agent())
return build_headers
def build_original_headers(referer):
"""Original image request headers
:param referer: headers need a last page referer
:return: build headers
"""
base_headers = {
'Accept': "image/webp,image/*,*/*;q=0.8",
'Accept-Encoding': "gzip, deflate, sdch",
'Accept-Language': _HEADERS_ACCEPT_LANGUAGE,
'Connection': _HEADERS_CONNECTION,
# must add referer, or server will return a damn http error 403, 404
# copy from javascript console network request headers of image
'Referer': referer, # request basic page
}
build_headers = dict(base_headers, **uc_user_agent())
return build_headers