-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathUtils.py
More file actions
executable file
·289 lines (222 loc) · 9.96 KB
/
Copy pathUtils.py
File metadata and controls
executable file
·289 lines (222 loc) · 9.96 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
# -*- coding: utf-8 -*-
import json
import ssl
from urllib.parse import quote
from Wappalyzer import Wappalyzer, WebPage
import mysql.connector
import urllib.request
# Shared utils
def dict_to_json(dictionary):
return json.dumps(dictionary, indent = 4)
def json_to_array(json_values):
return json.loads(json_values)
def json_to_dict(json_object):
return json.loads(json_object)
def contains_one_of(word, terms):
for term in terms:
if term in word:
return True
return False
def urlopen(url, timeout=60):
headers = {'User-Agent': 'Mozilla/5.0 Firefox/33.0'}
req = urllib.request.Request(url, None, headers)
return urllib.request.urlopen(req, timeout=timeout, context=ssl._create_unverified_context())
def tupes_to_dict(tup, di):
for a, b in tup:
di.setdefault(a, []).append(b)
return di
def extract_used_techs(url):
# webpage = urlopen(url, 60)
# content = webpage.read().decode('utf-8')
# headers = dict(webpage.getheaders())
# webpage = WebPage(url, html=content, headers=headers)#.new_from_url(url, verify=False)
webpage = WebPage.new_from_url(url, verify=False)
wappalyzer = Wappalyzer.latest()
return wappalyzer.analyze_with_versions_and_categories(webpage)
def extract_server(url):
techs = extract_used_techs(url)
for key in techs:
if 'Web servers' in techs[key]['categories']:
version = False
try:
version = techs[key]['versions'][techs[key]['categories'].index('Web servers')]
except:
version = False
return {'server': key, 'version': version}
return False
def extract_cms(url):
techs = techs = extract_used_techs(url)
for key in techs:
if 'CMS' in techs[key]['categories']:
version = False
try:
version = techs[key]['versions'][techs[key]['categories'].index('CMS')]
except:
version: False
return {'CMS': key, 'version': version}
return False
def accepted_url(target, nlink):
target_domain = target.replace('http://', '').replace('https://', '')
target_domain_without_www = target_domain.replace('www.', '')
page_url = nlink.split('/')[-1]
if (not nlink.endswith('.pdf') and (not nlink.endswith('.PDF')) and (not nlink.endswith('.rtf')) and (not nlink.endswith('.pptx')) and not nlink.endswith('.docx') and not nlink.endswith('.doc') and not nlink.endswith('.csv') and not nlink.endswith('.xlsx') and ('accounts.google.com') not in nlink) and (nlink.startswith('http://' + target_domain) or nlink.startswith('https://' + target_domain) or nlink.startswith('http://www.' + target_domain_without_www) or nlink.startswith('https://www.' + target_domain_without_www)) and ('.pdf' not in page_url) and ('download' not in nlink):
return True
return False
def read_file_items(file_path):
file = open(file_path, "r")
return file.readlines()
def fetch_all(query):
database = mysql.connector.connect(
host="localhost",
user="root",
passwd="root",
database="web_crawler"
)
query_cursor = database.cursor(buffered=True)
query_cursor.execute(query)
items = query_cursor.fetchall()
database.close()
return items
def fetch_one(query):
database = mysql.connector.connect(
host="localhost",
user="root",
passwd="root",
database="web_crawler"
)
query_cursor = database.cursor(buffered=True)
query_cursor.execute(query)
item = query_cursor.fetchone()
database.close()
return item
def commit_query(query):
database = mysql.connector.connect(
host="localhost",
user="root",
passwd="root",
database="web_crawler"
)
query_cursor = database.cursor()
query_cursor.execute(query)
database.commit()
database.close()
def find_category_by_name(category_name):
query = "SELECT * FROM categories WHERE name = '" + str(category_name) + "'"
return fetch_one(query)
def find_website_by_url(url):
query = "SELECT * FROM websites WHERE url = '" + str(url) + "'"
return fetch_one(query)
def find_website_by_category_name(category_name):
query = "SELECT websites.* FROM websites WHERE websites.category_id IN (SELECT categories.id FROM categories WHERE categories.name = '" + str(category_name) + "')"
return fetch_one(query)
def find_websites_by_category_name(category_name):
query = "SELECT websites.id, websites.url FROM websites WHERE websites.category_id IN (SELECT categories.id FROM categories WHERE categories.name = '" + str(category_name) + "')"
websites = fetch_all(query)
website_with_pages = []
for website in websites:
query = "SELECT url FROM pages WHERE website_id = '" + str(website[0]) + "'"
website_pages = fetch_all(query)
pages = []
for page in website_pages:
pages.append(page[0])
website_with_pages.append({'website': website[1], 'pages': pages})
return website_with_pages
def find_web_pages_by_category_name(category_name):
website = find_website_by_category_name(category_name)
if website:
query = "SELECT * FROM pages WHERE website_id = '" + str(website[0]) + "'"
return fetch_all(query)
return []
def insert_website_with_pages(category_name, website_url = '', pages = []):
category = find_category_by_name(category_name)
if not category:
query = "INSERT INTO categories (name) VALUES ('" + category_name + "')"
commit_query(query)
category = find_category_by_name(category_name)
website = find_website_by_url(website_url)
if not website:
query = "INSERT INTO websites (url, category_id) VALUES ('" + str(website_url) + "', '" + str(category[0]) + "')"
commit_query(query)
website = find_website_by_url(website_url)
for page in pages:
page = page.replace("'", "\\'").replace("\n", '')
print(page)
query = "INSERT INTO pages (url, website_id) VALUES ('" + str(page) + "', '" + str(website[0]) + "')"
commit_query(query)
def find_websites():
query = "SELECT * FROM websites"
return fetch_all(query)
def find_categories():
query = "SELECT * FROM categories"
return fetch_all(query)
def create_new_scan(label = ''):
query = "INSERT INTO scans (label) VALUES ('" + str(label) + "')"
commit_query(query)
return fetch_one("SELECT * FROM scans ORDER BY id DESC LIMIT 1")
def fin_rule_id_by_name(rule_name):
query = "SELECT * FROM rules WHERE name = '" + str(rule_name) + "'"
return fetch_one(query)[0]
def fin_page_id_by_url(page_url):
query = "SELECT * FROM pages WHERE url LIKE '" + str(page_url) + "'"
return fetch_one(query)[0]
def save_scan_page(scan_id, page_id):
query = "INSERT INTO `scan_page`(`page_id`, `scan_id`) VALUES ('" + str(page_id) + "','" + str(scan_id) + "')"
commit_query(query)
return fetch_one("SELECT * FROM scan_page ORDER BY id DESC LIMIT 1")
def save_scan_results(scan_page_id, rule_id, value):
is_secure = 1 if value else 0
query = "INSERT INTO `scan_page_results`(`scan_page_id`, `rule_id`, `is_secure`) VALUES ('" + str(scan_page_id) + "','" + str(rule_id) + "','" + str(is_secure) + "')"
commit_query(query)
return fetch_one("SELECT * FROM scan_page_results ORDER BY id DESC LIMIT 1")
def save_scan_results_by_scan_id(scan_id, results = []):
for result in results:
page_id = fin_page_id_by_url(result['url'])
scan_page = save_scan_page(scan_id, page_id)
for key, value in result['results'].items():
rule_id = fin_rule_id_by_name(key)
scan_result = save_scan_results(scan_page[0], rule_id, value)
def get_pourcentage_vuls_of_all_pages_by_scan(scan_id):
query = """
SELECT rules.name,
COUNT(scan_page_results.scan_page_id) as total_pages,
SUM(CASE WHEN scan_page_results.is_secure THEN 1 ELSE 0 END) as nbr_pages,
(SUM(CASE WHEN scan_page_results.is_secure THEN 1 ELSE 0 END) * 100 / COUNT(scan_page_results.scan_page_id)) as pourcentage
FROM scan_page_results, rules, scan_page
WHERE scan_page_results.rule_id = rules.id
AND scan_page.scan_id = '""" + str(scan_id) + """'
AND scan_page.id = scan_page_results.scan_page_id
GROUP BY scan_page_results.rule_id
"""
results = []
for item in fetch_all(query):
result_row = dict()
result_row['rule'] = item[0]
result_row['total_tested_pages'] = item[1]
result_row['total_founded_pages'] = item[2]
result_row['pourcentage'] = item[3]
results.append(result_row)
return results
def ge_pourcentage_vuls_of_category_by_scan(scan_id, category_id):
query = """
SELECT rules.name,
COUNT(scan_page_results.scan_page_id) as total_pages,
SUM(CASE WHEN scan_page_results.is_secure THEN 1 ELSE 0 END) as nbr_pages,
(SUM(CASE WHEN scan_page_results.is_secure THEN 1 ELSE 0 END) * 100 / COUNT(scan_page_results.scan_page_id)) as pourcentage
FROM scan_page_results, rules, scan_page, pages, websites
WHERE scan_page_results.rule_id = rules.id
AND scan_page.scan_id = '""" + str(scan_id) + """'
AND scan_page.id = scan_page_results.scan_page_id
AND scan_page.page_id = pages.id
AND pages.website_id = websites.id
AND websites.category_id = '""" + str(category_id) + """'
GROUP BY scan_page_results.rule_id
"""
results = []
for item in fetch_all(query):
result_row = dict()
result_row['rule'] = item[0]
result_row['total_tested_pages'] = item[1]
result_row['total_founded_pages'] = item[2]
result_row['pourcentage'] = item[3]
results.append(result_row)
return results