documentFingerprint.py 1.6 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455
  1. import hashlib
  2. import codecs
  3. from bs4 import BeautifulSoup
  4. import re
  5. def getHtmlText(sourceHtml):
  6. _soup = BeautifulSoup(sourceHtml,"lxml")
  7. list_a = _soup.find_all("a")
  8. for _a in list_a:
  9. _href = _a.attrs.get("href","")
  10. if _href.find("www.bidizhaobiao.com")>0:
  11. _a.decompose()
  12. # richText = _soup.find("div",attrs={"class":"richTextFetch"})
  13. # if richText is not None:
  14. # richText.decompose()
  15. _text = _soup.get_text()
  16. _text = re.sub("\s*",'',_text)
  17. # 删除“阅读量|点击量”等描述,避免同一站源重复爬取去重失败
  18. re_page_views = re.compile("(阅读|点击|浏览|访问|访客|阅览|查看|曝光)(量|数|次数|人数|人次)[::]?\d{1,8}")
  19. _text = re.sub(re_page_views,"",_text[:500],count=1) + _text[500:]
  20. if len(_text)==0:
  21. _text = str(_soup)
  22. return _text
  23. def getMD5(sourceHtml):
  24. if sourceHtml is not None and len(sourceHtml)>0:
  25. _text = getHtmlText(sourceHtml)
  26. if isinstance(_text,str):
  27. bs = _text.encode()
  28. elif isinstance(_text,bytes):
  29. bs = _text
  30. else:
  31. return ""
  32. md5 = hashlib.md5()
  33. md5.update(bs)
  34. return md5.hexdigest()
  35. return ""
  36. def getFingerprint(sourceHtml):
  37. md5 = getMD5(sourceHtml)
  38. if md5!="":
  39. _fingerprint = "md5=%s"%(md5)
  40. else:
  41. _fingerprint = ""
  42. return _fingerprint
  43. if __name__=="__main__":
  44. sourceHtml = codecs.open("C:\\Users\\\Administrator\\Desktop\\2.html","rb",encoding="utf8").read()
  45. # sourceHtml = "abcddafafffffffffffffffffffffffff你"
  46. print(getFingerprint(sourceHtml))