Large file segmentation, naming script-Python

Log file division and naming#

At work, I often receive log files provided by test students and client students, many of which are hundreds of M to G. After all, the amount of logs generated by stress testing in one night is still considerable, xDxD, so it is inevitable that you need to correct The log is divided, and usually the problem needs to be located at the time point. Therefore, it is best to name the divided log file with the start and end time points of the log in the file. This is the most intuitive to use. Let’s share two scripts for you. Divide, name, hope to provide a little help to everyone;

Large file segmentation##

usage:

  1. python split_big_file.py
  2. Enter the full path name of the file
  3. Enter the desired number of lines in each small file after splitting
  4. Just wait.

code show as below:

	# -*- coding:utf-8-*-import os,re,shutil
	import platform

	sys_name = platform.system().lower()
	SPLIT_CHAR ='\\'if sys_name.find('windows')!=-1else'/'print('input big files`s path:')
	_path = raw_input()
	names = []
	pathes = []
	if os.path.isfile(_path):
		print('is file')
		names.append(_path)
	else:
		print('is nothing')
	'''
	elif os.path.isdir(_path):
		print('This is dir')
		pathes = os.listdir(_path)
		print('pathes='+str(pathes))
		for i in range(len(pathes)):
			fullpath = _path+SPLIT_CHAR+pathes[i]
			print('fullpath='+fullpath)
			if os.path.isfile(fullpath):
				names.append(fullpath)
				files.append(open(fullpath).read().split('\n'))
	'''
		
	print(len(names))

	line_num = int(raw_input('every file`line num = '))print('line number='+str(line_num))for i inrange(len(names)):
		_name = names[i]
		ori_name = _name.split(SPLIT_CHAR)[len(_name.split(SPLIT_CHAR))-1]
		dir_name = _name.replace(ori_name,'DIR_'+ori_name)
		dir_name = dir_name.replace('.','_')
		print ori_name
		print dir_name
		os.system('mkdir '+dir_name)
		count =1
		print 'Processed:'+str(count)+'Row'
		part_file =open(dir_name+SPLIT_CHAR+str(0)+'.part.txt','w')withopen(_name,'rb')as f:for line in f:if count%line_num ==0:
			    part_file.close()
			    part_file =open(dir_name+SPLIT_CHAR+str(int(count/line_num))+'.part.txt','w')
			part_file.write(line+'\n')
			count+=1if count%100000==0:
			    print 'Processed:'+str(count)+'Row'
		print 'Processed:'+str(count)+'Row'
		os.system('python ./get_name_logfile.py '+dir_name)

The file is renamed according to the start and end line timestamp##

usage:

You can select a file or folder as a parameter. If it is a folder, each file in the folder will be processed (not recursively to the files in the folder under the folder);

code show as below:

	# -*- coding:utf-8-*-import os,re,shutil
	import sys
	import platform

	sys_name = platform.system().lower()
	SPLIT_CHAR ='\\'if sys_name.find('windows')!=-1else'/'

	_path = sys.argv[1]
	names =[]
	files =[]
	pathes =[]if os.path.isfile(_path):print('is file')
		names[0]= _path
	elif os.path.isdir(_path):print('This is dir')
		pathes = os.listdir(_path)print('pathes='+str(pathes))for i inrange(len(pathes)):
			fullpath = _path+SPLIT_CHAR+pathes[i]print('fullpath='+fullpath)if os.path.isfile(fullpath):
				names.append(fullpath)else:print('is nothing')print(len(names))

	#Date format: 05-2618:20:42.093	r'\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}.\d{3}'
	#	
	#	05-2618:20:43.093:r'\d{2}-\d{2} {1,}\d{2}:\d{2}:\d{2}.\d{1,10}'

	date_reg = r'\d{2}-\d{2} {1,}\d{2}:\d{2}:\d{2}.\d{1,10}'
	time_reg = r'\d{2}:\d{2}:\d{2}.\d{1,10}'for i inrange(len(names)):
		_name = names[i]print('name='+_name)
		#head tries to find the date in 10 rows
		head_len =10
		start_time ='(start_time-'
		_file_ =open(_name,'rb')
		reads = _file_.read()
		_file = reads.split('\n')iflen(_file)/2<10:
			head_len =len(_file)/2for j inrange(head_len):
			res = re.search(date_reg, _file[j])if res!=None and res.group(0)!=None:
				start_time = res.group(0)print('start_time='+start_time)break
		# tail
		tail_len =len(_file)-head_len
		end_time ='-end_time)'for j inrange(len(_file)-1,tail_len-1,-1):
			res = re.search(time_reg, _file[j])if res!=None and res.group(0)!=None:
				end_time = res.group(0)print('end_time='+end_time)break
		_file_.close()
		ori_name = _name.split(SPLIT_CHAR)[len(_name.split(SPLIT_CHAR))-1]print('ori_name='+ori_name)
		new_name = start_time.replace(':','-')+'__'+end_time.replace(':','-')+os.path.splitext(ori_name)[1]print('new_name='+new_name)print("copy %s %s"%(_name, _name.replace(ori_name,new_name)))
		#os.system("copy %s %s"%(_name, _name.replace(ori_name,new_name)))
		shutil.copy(_name,_name.replace(ori_name,new_name))
		os.system("rm -rf "+_name)

Recommended Posts

Large file segmentation, naming script-Python