AccountArticleRank.py 11 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335
  1. """
  2. @author: luojunhui
  3. """
  4. import time
  5. from uuid import uuid4
  6. from tqdm import tqdm
  7. from applications.aliyunLog import AliyunArticleLog
  8. from applications.functions import ArticleRank, title_sim_v2_by_list
  9. from applications.pipeline import LongArticlesPipeline
  10. def deduplication(rank1, rank2, rank3):
  11. """
  12. 标题相似度去重
  13. :return:
  14. """
  15. dup_list = []
  16. final_result = []
  17. for item_list in [rank1, rank2, rank3]:
  18. result = []
  19. if item_list:
  20. for item in item_list:
  21. title = item['title']
  22. if title_sim_v2_by_list(title, dup_list):
  23. # print("标题重复,已经过滤\t", title)
  24. continue
  25. else:
  26. result.append(item)
  27. dup_list.append(title)
  28. final_result.append(result)
  29. return final_result[0], final_result[1], final_result[2]
  30. class AccountArticleRank(object):
  31. """
  32. 文章排序
  33. """
  34. def __init__(self, params, mysql_client):
  35. """
  36. :param params: 请求参数
  37. :param mysql_client: 数据库链接池
  38. """
  39. self.filter_list = None
  40. self.publishArticleList = None
  41. self.publishNum = None
  42. self.strategy = None
  43. self.ghId = None
  44. self.accountName = None
  45. self.accountId = None
  46. self.params = params
  47. self.mysql_client = mysql_client
  48. self.request_id = "alg-{}-{}".format(uuid4(), int(time.time()))
  49. self.logger = AliyunArticleLog(request_id=self.request_id, alg="ArticleRank")
  50. self.pipeline = LongArticlesPipeline()
  51. def filter(self):
  52. """
  53. 过滤器
  54. """
  55. self.publishArticleList = []
  56. self.filter_list = []
  57. history_title_dict = self.pipeline.history_title(account_nickname=self.accountName)
  58. for item in tqdm(self.params['publishArticleList']):
  59. flag = self.pipeline.deal(item, self.accountName, history_title_dict)
  60. if flag:
  61. item['filterReason'] = flag['filterReason']
  62. self.filter_list.append(item)
  63. else:
  64. self.publishArticleList.append(item)
  65. print("过滤完成")
  66. async def check_params(self):
  67. """
  68. 校验参数
  69. :return:
  70. """
  71. try:
  72. self.accountId = self.params["accountId"]
  73. self.accountName = self.params["accountName"]
  74. self.ghId = self.params["ghId"]
  75. self.strategy = self.params["strategy"]
  76. self.publishNum = self.params["publishNum"]
  77. print("开始校验参数")
  78. self.filter()
  79. self.logger.log(
  80. code="1001",
  81. msg="参数校验成功",
  82. data=self.params
  83. )
  84. return None
  85. except Exception as e:
  86. response = {
  87. "msg": "params error",
  88. "info": "params check failed, params : {} is not correct".format(e),
  89. "code": 0,
  90. }
  91. self.logger.log(
  92. code="1002",
  93. msg="参数校验失败--{}".format(e),
  94. data=self.params
  95. )
  96. return response
  97. async def basic_rank(self):
  98. """
  99. 基础排序
  100. :return:
  101. """
  102. # 第一步把所有文章标题分为3组
  103. article_list1_ori = [i for i in self.publishArticleList if "【1】" in i['producePlanName']]
  104. article_list2_ori = [i for i in self.publishArticleList if "【2】" in i['producePlanName']]
  105. article_list3_ori = [i for i in self.publishArticleList if
  106. not i in article_list1_ori and not i in article_list2_ori]
  107. # # 全局去重,保留优先级由 L1 --> L2 --> L3
  108. # hash_map = {}
  109. #
  110. # article_list1 = []
  111. # for i in article_list1_ori:
  112. # title = i['title']
  113. # if hash_map.get(title):
  114. # continue
  115. # else:
  116. # article_list1.append(i)
  117. # hash_map[title] = 1
  118. #
  119. # article_list2 = []
  120. # for i in article_list2_ori:
  121. # title = i['title']
  122. # if hash_map.get(title):
  123. # continue
  124. # else:
  125. # article_list2.append(i)
  126. # hash_map[title] = 2
  127. #
  128. # article_list3 = []
  129. # for i in article_list3_ori:
  130. # title = i['title']
  131. # if hash_map.get(title):
  132. # continue
  133. # else:
  134. # article_list3.append(i)
  135. # hash_map[title] = 1
  136. # 第二步对article_list1, article_list3按照得分排序, 对article_list2按照播放量排序
  137. if article_list1_ori:
  138. rank1 = ArticleRank().rank(
  139. account_list=[self.accountName],
  140. text_list=[i['title'] for i in article_list1_ori]
  141. )
  142. score_list1 = rank1[self.accountName]['score_list']
  143. ranked_1 = []
  144. for index, value in enumerate(score_list1):
  145. obj = article_list1_ori[index]
  146. obj['score'] = value + 1000
  147. ranked_1.append(obj)
  148. ranked_1 = sorted(ranked_1, key=lambda x: x['score'], reverse=True)
  149. else:
  150. ranked_1 = []
  151. # rank2
  152. if article_list2_ori:
  153. for item in article_list2_ori:
  154. item['score'] = 100
  155. ranked_2 = sorted(article_list2_ori, key=lambda x: x['crawlerViewCount'], reverse=True)
  156. else:
  157. ranked_2 = []
  158. # rank3
  159. if article_list3_ori:
  160. rank3 = ArticleRank().rank(
  161. account_list=[self.accountName],
  162. text_list=[i['title'] for i in article_list3_ori]
  163. )
  164. score_list3 = rank3[self.accountName]['score_list']
  165. ranked_3 = []
  166. for index, value in enumerate(score_list3):
  167. obj = article_list3_ori[index]
  168. obj['score'] = value
  169. ranked_3.append(obj)
  170. ranked_3 = sorted(ranked_3, key=lambda x: x['score'], reverse=True)
  171. else:
  172. ranked_3 = []
  173. self.logger.log(
  174. code="1004",
  175. msg="排序完成",
  176. data={
  177. "rank1": ranked_1,
  178. "rank2": ranked_2,
  179. "rank3": ranked_3
  180. }
  181. )
  182. return ranked_1, ranked_2, ranked_3
  183. async def rank_v1(self):
  184. """
  185. Rank Version 1
  186. :return:
  187. """
  188. print("开始排序")
  189. try:
  190. ranked_1_d, ranked_2_d, ranked_3_d = await self.basic_rank()
  191. ranked_1, ranked_2, ranked_3 = deduplication(ranked_1_d, ranked_2_d, ranked_3_d)
  192. print("去重成功")
  193. try:
  194. L = []
  195. if ranked_1:
  196. L.append(ranked_1[0])
  197. if ranked_2:
  198. L.append(ranked_2[0])
  199. else:
  200. if ranked_2:
  201. if len(ranked_2) > 1:
  202. for i in ranked_2[:2]:
  203. L.append(i)
  204. else:
  205. L.append(ranked_2[0])
  206. for item in ranked_3:
  207. L.append(item)
  208. result = {
  209. "accountId": self.accountId,
  210. "accountName": self.accountName,
  211. "ghId": self.ghId,
  212. "strategy": self.strategy,
  213. "publishNum": self.publishNum,
  214. "rank_list": L[:self.publishNum],
  215. "filter_list": self.filter_list
  216. }
  217. self.logger.log(
  218. code=1006,
  219. msg="rank successfully",
  220. data=result
  221. )
  222. response = {"status": "Rank Success", "data": result, "code": 1}
  223. except Exception as e:
  224. result = {
  225. "accountId": self.accountId,
  226. "accountName": self.accountName,
  227. "ghId": self.ghId,
  228. "strategy": self.strategy,
  229. "publishNum": self.publishNum,
  230. "rank_list": self.publishArticleList[: self.publishNum],
  231. "filter_list": self.filter_list
  232. }
  233. self.logger.log(
  234. code=1007,
  235. msg="rank failed because of {}".format(e),
  236. data=result
  237. )
  238. print("排序成功")
  239. response = {"status": "Rank Fail", "data": result, "code": 1}
  240. return response
  241. except:
  242. result = {"code": 2, "info": "account is not exist"}
  243. return result
  244. async def rank_v2(self):
  245. """
  246. Rank Version 2
  247. :return:
  248. """
  249. return await self.rank_v1()
  250. async def rank_v3(self):
  251. """
  252. Rank Version 3
  253. :return:
  254. """
  255. return await self.rank_v1()
  256. async def rank_v4(self):
  257. """
  258. Rank Version 4
  259. :return:
  260. """
  261. return await self.rank_v1()
  262. async def rank_v5(self):
  263. """
  264. Rank Version 5
  265. :return:
  266. """
  267. return await self.rank_v1()
  268. async def choose_strategy(self):
  269. """
  270. 选择排序策略
  271. :return:
  272. """
  273. match self.strategy:
  274. case "ArticleRankV1":
  275. self.logger.log(
  276. code="1003",
  277. msg="命中排序策略1"
  278. )
  279. return await self.rank_v1()
  280. case "ArticleRankV2":
  281. self.logger.log(
  282. code="1003",
  283. msg="命中排序策略2"
  284. )
  285. return await self.rank_v2()
  286. case "ArticleRankV3":
  287. self.logger.log(
  288. code="1003",
  289. msg="命中排序策略3"
  290. )
  291. return await self.rank_v3()
  292. case "ArticleRankV4":
  293. self.logger.log(
  294. code="1003",
  295. msg="命中排序策略4"
  296. )
  297. return await self.rank_v4()
  298. case "ArticleRankV5":
  299. self.logger.log(
  300. code="1003",
  301. msg="命中排序策略5"
  302. )
  303. return await self.rank_v5()
  304. async def deal(self):
  305. """
  306. Deal Function
  307. :return:
  308. """
  309. error_params = await self.check_params()
  310. if error_params:
  311. return error_params
  312. else:
  313. print("参数校验成功")
  314. return await self.choose_strategy()