change with tmp files
This commit is contained in:
parent
9f87f38347
commit
699cecad4f
@ -23,9 +23,9 @@ def errorRevert(logger, revert, tmp):
|
|||||||
logger.error("Error revert, because number files tmp is incompatible with parallel number")
|
logger.error("Error revert, because number files tmp is incompatible with parallel number")
|
||||||
exit(1)
|
exit(1)
|
||||||
|
|
||||||
def change(index, number, args, logger):
|
def change(index, number, args, logger, tmp, revert):
|
||||||
changeWp = WPChange(logger=logger, index_name=index, number_thread=number)
|
changeWp = WPChange(logger=logger, index_name=index, number_thread=number, tmp=tmp)
|
||||||
changeWp.fromDirectory(args.directory)
|
changeWp.fromDirectory(args.directory, revert)
|
||||||
|
|
||||||
del changeWp
|
del changeWp
|
||||||
|
|
||||||
@ -333,7 +333,8 @@ if __name__ == '__main__':
|
|||||||
if len(args.directory) > 0:
|
if len(args.directory) > 0:
|
||||||
try:
|
try:
|
||||||
with futures.ThreadPoolExecutor(max_workers=int(args.parallel)) as ex:
|
with futures.ThreadPoolExecutor(max_workers=int(args.parallel)) as ex:
|
||||||
wait_for = [ ex.submit(change, i, args.parallel, args, logger) for i in range(0, int(args.parallel)) ]
|
errorRevert(logger, args.revert, args.tmp)
|
||||||
|
wait_for = [ ex.submit(change, i, args.parallel, args, logger, args.tmp, args.revert) for i in range(0, int(args.parallel)) ]
|
||||||
except Exception as err:
|
except Exception as err:
|
||||||
logger.error("Thread error for remove : {0}".format(err))
|
logger.error("Thread error for remove : {0}".format(err))
|
||||||
if len(args.file) > 0:
|
if len(args.file) > 0:
|
||||||
|
@ -4,11 +4,13 @@ import requests, os, logging, re, json
|
|||||||
|
|
||||||
class WPChange:
|
class WPChange:
|
||||||
# Constructor
|
# Constructor
|
||||||
def __init__(self, index_name=1, number_thread=1, logger=None, parser="html.parser"):
|
def __init__(self, index_name=1, number_thread=1, logger=None, parser="html.parser", tmp="/tmp/import_export_canablog"):
|
||||||
self._name = "Thread-{0}".format(index_name)
|
self._name = "Thread-{0}".format(index_name)
|
||||||
self._logger = logger
|
self._logger = logger
|
||||||
self._number_thread = number_thread
|
self._number_thread = number_thread
|
||||||
self._parser = parser
|
self._parser = parser
|
||||||
|
self._tmp = tmp
|
||||||
|
self._index_name = index_name
|
||||||
|
|
||||||
# Destructor
|
# Destructor
|
||||||
def __del__(self):
|
def __del__(self):
|
||||||
@ -38,18 +40,61 @@ class WPChange:
|
|||||||
|
|
||||||
## From directory
|
## From directory
|
||||||
|
|
||||||
def fromDirectory(self, directory="", number_thread=1, max_thread=1):
|
|
||||||
|
def fromDirectory(self, directory="", revert=False):
|
||||||
|
self._directory = directory
|
||||||
directory = "{0}/archives".format(directory)
|
directory = "{0}/archives".format(directory)
|
||||||
directories = self._getDirectories([], "{0}".format(directory))
|
directories = self._getDirectories([], "{0}".format(directory))
|
||||||
if len(directories) > 0:
|
if len(directories) > 0:
|
||||||
files = self._getFiles(directories)
|
files = self._getFiles(directories)
|
||||||
self.fromFile(files, number_thread, max_thread)
|
if revert is False:
|
||||||
|
self._tmpFiles(files=files, number_thread=self._index_name, max_thread=self._number_thread)
|
||||||
|
self._fromFileTmp()
|
||||||
else:
|
else:
|
||||||
self._logger.error("{0} : No files for {1}".format(self._name, directory))
|
self._logger.error("{0} : No files for {1}".format(self._name, directory))
|
||||||
|
|
||||||
|
|
||||||
|
def fromFile(self, files=[]):
|
||||||
|
for i in range(0, len(files)):
|
||||||
|
if os.path.exists(files[i]):
|
||||||
|
self._logger.info("{0} : ({1}/{2}) File is being processed : {3}".format(self._name, i+1, len(files), files[i]))
|
||||||
|
self._change(files[i])
|
||||||
|
|
||||||
|
|
||||||
# Private method
|
# Private method
|
||||||
|
|
||||||
|
def _fromFileTmp(self):
|
||||||
|
try:
|
||||||
|
with open("{0}/{1}.json".format(self._tmp, self._name)) as file:
|
||||||
|
files = json.loads(file.read())
|
||||||
|
self._logger.debug("{0} : size of webpage : {1}".format(self._name, len(files)))
|
||||||
|
for i in range(0, len(files)):
|
||||||
|
if os.path.exists(files[i]):
|
||||||
|
self._logger.info("{0} : ({1}/{2}) File is being processed : {3}".format(self._name, i+1, len(files), files[i]))
|
||||||
|
self._change(files[i])
|
||||||
|
except Exception as ex:
|
||||||
|
self._logger.error("{0} : Read file json from tmp : {1}".format(self._name, ex))
|
||||||
|
|
||||||
|
|
||||||
|
def _tmpFiles(self, files=[], number_thread=1, max_thread=1):
|
||||||
|
print()
|
||||||
|
divFiles = int(len(files) / int(max_thread))
|
||||||
|
currentRangeFiles = int(divFiles * (int(number_thread)+1))
|
||||||
|
firstRange = int(currentRangeFiles - divFiles)
|
||||||
|
self._logger.debug("{0} : index : {1}".format(self._name,number_thread))
|
||||||
|
|
||||||
|
self._logger.debug("{0} : first range : {1}".format(self._name,firstRange))
|
||||||
|
self._logger.debug("{0} : last range : {1}".format(self._name,currentRangeFiles))
|
||||||
|
webpage = []
|
||||||
|
for i in range(firstRange, currentRangeFiles):
|
||||||
|
webpage.append(files[i])
|
||||||
|
|
||||||
|
try:
|
||||||
|
string_webpage = json.dumps(webpage)
|
||||||
|
open("{0}/{1}.json".format(self._tmp, self._name), "wt").write(string_webpage)
|
||||||
|
except Exception as ex:
|
||||||
|
self._logger.error("{0} : Error for writing webpage : {1}".format(self._name, ex))
|
||||||
|
|
||||||
## Get all files
|
## Get all files
|
||||||
|
|
||||||
def _getFiles(self, item):
|
def _getFiles(self, item):
|
||||||
|
Loading…
x
Reference in New Issue
Block a user