only parse html files
This commit is contained in:
1 parent
8adc30d7ae
commit
631dade746
1 file changed
+6
Regular → Executable
+6
@@ -1,3 +1,5 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
import argparse
|
||||
import urllib.parse
|
||||
import urllib3
|
||||
@@ -55,6 +57,10 @@ class Crawler:
|
||||
|
||||
self.visited.add(url)
|
||||
res = self.request(url)
|
||||
content_type = res.headers.get("Content-Type", None)
|
||||
if "text/html" not in content_type.lower().split(";"):
|
||||
continue
|
||||
|
||||
urls = self.collect_urls(res.text)
|
||||
for url in urls:
|
||||
parts = urllib.parse.urlparse(url)
|
||||
|
||||
Reference in new issue
Block a user