web_scrap/lib/WPChange.py

129 lines
5.1 KiB
Python
Raw Normal View History

2023-06-05 23:46:57 +02:00
from bs4 import BeautifulSoup
from urllib.parse import urlparse
import requests, os, logging, re, json
2023-06-10 01:58:08 +02:00
class WPChange:
2023-06-05 23:46:57 +02:00
# Constructor
2023-06-06 00:22:16 +02:00
def __init__(self, index_name=1, number_thread=1, logger=None, parser="html.parser"):
2023-06-05 23:46:57 +02:00
self._name = "Thread-{0}".format(index_name)
self._logger = logger
self._number_thread = number_thread
2023-06-06 00:22:16 +02:00
self._parser = parser
2023-06-05 23:46:57 +02:00
# Destructor
def __del__(self):
print("{0} : Import finished".format(self._name))
# Public method
## from file
2023-06-06 00:22:16 +02:00
2023-06-05 23:46:57 +02:00
def fromFile(self, files=[], number_thread=1, max_thread=1):
divFiles = int(len(files) / max_thread)
2023-06-11 20:24:22 +02:00
currentRangeFiles = int(divFiles * (number_thread))
2023-06-05 23:46:57 +02:00
firstRange = int(currentRangeFiles - divFiles)
self._logger.debug("{0} : index : {1}".format(self._name,number_thread))
2023-06-11 20:24:22 +02:00
2023-06-05 23:46:57 +02:00
self._logger.debug("{0} : first range : {1}".format(self._name,firstRange))
self._logger.debug("{0} : last range : {1}".format(self._name,currentRangeFiles))
for i in range(firstRange, currentRangeFiles):
2023-06-11 20:24:22 +02:00
2023-06-05 23:46:57 +02:00
if os.path.exists(files[i]):
self._logger.info("{0} : ({1}/{2}) File is being processed : {3}".format(self._name, i+1, currentRangeFiles + 1, files[i]))
2023-06-06 00:22:16 +02:00
self._change(files[i])
2023-06-05 23:46:57 +02:00
## From directory
def fromDirectory(self, directory="", number_thread=1, max_thread=1):
directory = "{0}/archives".format(directory)
directories = self._getDirectories([], "{0}".format(directory))
if len(directories) > 0:
files = self._getFiles(directories)
self.fromFile(files, number_thread, max_thread)
else:
self._logger.error("{0} : No files for {1}".format(self._name, directory))
# Private method
## Get all files
def _getFiles(self, item):
files = []
for i in item:
for j in os.listdir(i):
if os.path.isfile("{0}/{1}".format(i, j)):
files.append("{0}/{1}".format(i, j))
return files
## Get directories
def _getDirectories(self, subdirectory, item):
sub = subdirectory
for i in os.listdir(item):
if os.path.isdir("{0}/{1}".format(item, i)):
sub.append("{0}/{1}".format(item, i))
subdirectory = self._getDirectories(sub, "{0}/{1}".format(item, i))
return subdirectory
2023-06-06 00:22:16 +02:00
## Change path img file
def _change(self, file):
2023-06-13 00:46:18 +02:00
ext_img = ["png", "svg", "gif", "jpg", "jpeg"]
2023-06-11 20:24:22 +02:00
try:
with open(file, 'r') as f:
content = f.read()
soup = BeautifulSoup(content, self._parser)
img = soup.find_all("img")
for i in img:
src = i.get("src")
o = urlparse(src)
if len(o.netloc) > 0:
2023-06-13 00:38:34 +02:00
self._logger.info("{0} : Change source image {1} /img/{2}/{3}".format(self._name, src, o.netloc, o.path))
content = content.replace(src, "/img/{0}/{1}".format(o.netloc, o.path))
script = soup.find_all("script", {"type": "text/javascript"})
for i in script:
src = i.get("src")
if src is not None:
o = urlparse(src)
if len(o.netloc) > 0:
self._logger.info("{0} : Change source js {1} /dists/js/{2}/{3}".format(self._name, src, o.netloc, o.path))
content = content.replace(src, "/dists/js/{0}/{1}".format(o.netloc, o.path))
link = soup.find_all("link", {"rel": "stylesheet"})
for i in link:
href = i.get("href")
if href is not None:
o = urlparse(href)
if len(o.netloc) > 0:
self._logger.info("{0} : Change source css {1} /dists/css/{2}/{3}".format(self._name, href, o.netloc, o.path))
content = content.replace(href, "/dists/css/{0}/{1}".format(o.netloc, o.path))
2023-06-13 00:46:18 +02:00
a = soup.find_all("a", {"target": "_blank"})
for i in a:
href = i.get("href")
if href is not None:
o = urlparse(href)
if len(o.netloc) > 0:
ext = o.path.split(".")[len(o.path.split("."))-1]
if ext in ext_img:
self._logger.info("{0} : Change a img {1} /img/{2}/{3}".format(self._name, href, o.netloc, o.path))
content = content.replace(href, "/img/{0}/{1}".format(o.netloc, o.path))
2023-06-11 20:24:22 +02:00
try:
with open(file, "w") as f:
self._logger.info("{0} : File write : {1}".format(self._name, file))
f.write(content)
except Exception as ex:
self._logger.error("{0} : Error for write file {1} : {2}".format(self._name, file, ex))
except Exception as ex:
self._logger.error("{0} : Error for read file {1} : {2}".format(self._name, file, ex))
2023-06-06 00:22:16 +02:00