python获取抖音直播列表 python怎么爬取直播视频

转载

技术笔耕者 2024-06-18 10:28:17

文章标签 python获取抖音直播列表 python ide html Python 文章分类 Python 后端开发

这里需要说明的就只，有的图片资源并不是url链接，是data:image格式，这里需要转换一下存储！

def getResourceUrlList(url ,isImage, isAudio, isVideo):
	global imgType_list, audioType_list, videoType_list
	imageUrlList = []
	audioUrlList = []
	videoUrlList = []
 
	url = url.rstrip().rstrip('/')
	htmlStr = str(requestsDataBase(url))
	# print(htmlStr)
	
	Wopen = open('reptileHtml.txt','w')
	Wopen.write(htmlStr)
	Wopen.close()
 
	Ropen = open('reptileHtml.txt','r')
	imageUrlList = []
 
	for line in Ropen:
		line = line.replace("'", '"')
		segmenterStr = '"'
		if "'" in line:
			segmenterStr = "'"
 
		lineList = line.split(segmenterStr)
		for partLine in lineList:
			if isImage == True:
				# 查找图片
				if 'data:image' in partLine:
					base64List = partLine.split('base64,')
					imgData = base64.urlsafe_b64decode(base64List[-1] + '=' * (4 - len(base64List[-1]) % 4))
					base64ImgType = base64List[0].split('/')[-1].rstrip(';')
					imageName = zfjTools.getTimestamp() + '.' + base64ImgType
					imageUrlList.append(imageName + '$==$' + base64ImgType)
 
				# 查找图片
				for imageType in imgType_list:
					if imageType in partLine:
						imgUrl = partLine[:partLine.find(imageType) + len(imageType)].split(segmenterStr)[-1]
 
						# 修复URL
						imgUrl = repairUrl(imgUrl, url)
 
						sizeType = '_{' + 'size' + '}'
						if sizeType in imgUrl:
							imgUrl = imgUrl.replace(sizeType, '')
 
						imgUrl = imgUrl.strip()
 
						if imgUrl.startswith('http://') or imgUrl.startswith('https://') and imgUrl not in imageUrlList:
							imageUrlList.append(imgUrl)
						else:
							imgUrl = ''
 
			if isAudio == True:
				# 查找音频
				for audioType in audioType_list:
					if audioType in partLine or audioType.lower() in partLine:
						audioType = audioType.lower() if audioType.lower() in partLine else audioType
						audioUrl = partLine[:partLine.find(audioType) + len(audioType)].split(segmenterStr)[-1]
 
						# 修复URL
						audioUrl = repairUrl(audioUrl, url)
 
						if audioUrl.startswith('http://') or audioUrl.startswith('https://') and audioUrl not in audioUrlList:
							audioUrlList.append(audioUrl)
						else:
							audioUrl = ''
 
			if isVideo == True:
				# 查找视频
				for videoType in videoType_list:
					if videoType in partLine or videoType.lower() in partLine:
						videoType = videoType.lower() if videoType.lower() in partLine else videoType
						videoUrl = partLine[:partLine.find(videoType) + len(videoType)].split(segmenterStr)[-1]
 
						# 修复URL
						videoUrl = repairUrl(videoUrl, url)
 
						if videoUrl.startswith('http://') or videoUrl.startswith('https://') or videoUrl.startswith('ed2k://') or videoUrl.startswith('magnet:?') or videoUrl.startswith('ftp://') and videoUrl not in videoUrlList:
							videoUrlList.append(videoUrl)
						else:
							videoUrl = ''
 
	return (imageUrlList, audioUrlList, videoUrlList)
复制代码

爬取自定义节点

# 统配节点爬取
def getNoteInfors(url, fatherNode, childNode):
	url = url.rstrip().rstrip('/')
	htmlStr = requestsDataBase(url)
	
	Wopen = open('reptileHtml.txt','w')
	Wopen.write(htmlStr)
	Wopen.close()

	html_etree = etree.HTML(htmlStr)

	dataArray = []

	if html_etree != None:
		nodes_list = html_etree.xpath(fatherNode)
		for k_value in nodes_list:
			partValue = k_value.xpath(childNode)
			if len(partValue) > 0:
				dataArray.append(partValue[0])

	return dataArray
复制代码