AccountArticleRank.py 11 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338
  1. """
  2. @author: luojunhui
  3. """
  4. import random
  5. import time
  6. from uuid import uuid4
  7. from tqdm import tqdm
  8. from applications.aliyunLog import AliyunArticleLog
  9. from applications.functions import ArticleRank, title_sim_v2_by_list
  10. from applications.pipeline import LongArticlesPipeline
  11. def deduplication(rank1, rank2, rank3):
  12. """
  13. 标题相似度去重
  14. :return:
  15. """
  16. dup_list = []
  17. final_result = []
  18. for item_list in [rank1, rank2, rank3]:
  19. result = []
  20. if item_list:
  21. for item in item_list:
  22. title = item['title']
  23. if title_sim_v2_by_list(title, dup_list):
  24. # print("标题重复,已经过滤\t", title)
  25. continue
  26. else:
  27. result.append(item)
  28. dup_list.append(title)
  29. final_result.append(result)
  30. return final_result[0], final_result[1], final_result[2]
  31. class AccountArticleRank(object):
  32. """
  33. 文章排序
  34. """
  35. def __init__(self, params, mysql_client):
  36. """
  37. :param params: 请求参数
  38. :param mysql_client: 数据库链接池
  39. """
  40. self.filter_list = None
  41. self.publishArticleList = None
  42. self.publishNum = None
  43. self.strategy = None
  44. self.ghId = None
  45. self.accountName = None
  46. self.accountId = None
  47. self.params = params
  48. self.mysql_client = mysql_client
  49. self.request_id = "alg-{}-{}".format(uuid4(), int(time.time()))
  50. self.logger = AliyunArticleLog(request_id=self.request_id, alg="ArticleRank")
  51. self.pipeline = LongArticlesPipeline()
  52. def filter(self):
  53. """
  54. 过滤器
  55. """
  56. self.publishArticleList = []
  57. self.filter_list = []
  58. history_title_dict = self.pipeline.history_title(account_nickname=self.accountName)
  59. for item in tqdm(self.params['publishArticleList']):
  60. flag = self.pipeline.deal(item, self.accountName, history_title_dict)
  61. if flag:
  62. item['filterReason'] = flag['filterReason']
  63. self.filter_list.append(item)
  64. else:
  65. self.publishArticleList.append(item)
  66. print("过滤完成")
  67. async def check_params(self):
  68. """
  69. 校验参数
  70. :return:
  71. """
  72. try:
  73. self.accountId = self.params["accountId"]
  74. self.accountName = self.params["accountName"]
  75. self.ghId = self.params["ghId"]
  76. self.strategy = self.params["strategy"]
  77. self.publishNum = self.params["publishNum"]
  78. print("开始校验参数")
  79. self.filter()
  80. self.logger.log(
  81. code="1001",
  82. msg="参数校验成功",
  83. data=self.params
  84. )
  85. return None
  86. except Exception as e:
  87. response = {
  88. "msg": "params error",
  89. "info": "params check failed, params : {} is not correct".format(e),
  90. "code": 0,
  91. }
  92. self.logger.log(
  93. code="1002",
  94. msg="参数校验失败--{}".format(e),
  95. data=self.params
  96. )
  97. return response
  98. async def basic_rank(self):
  99. """
  100. 基础排序
  101. :return:
  102. """
  103. # 第一步把所有文章标题分为3组
  104. article_list1_ori = [i for i in self.publishArticleList if "【1】" in i['producePlanName']]
  105. article_list2_ori = [i for i in self.publishArticleList if "【2】" in i['producePlanName']]
  106. article_list3_ori = [i for i in self.publishArticleList if
  107. not i in article_list1_ori and not i in article_list2_ori]
  108. # # 全局去重,保留优先级由 L1 --> L2 --> L3
  109. # hash_map = {}
  110. #
  111. # article_list1 = []
  112. # for i in article_list1_ori:
  113. # title = i['title']
  114. # if hash_map.get(title):
  115. # continue
  116. # else:
  117. # article_list1.append(i)
  118. # hash_map[title] = 1
  119. #
  120. # article_list2 = []
  121. # for i in article_list2_ori:
  122. # title = i['title']
  123. # if hash_map.get(title):
  124. # continue
  125. # else:
  126. # article_list2.append(i)
  127. # hash_map[title] = 2
  128. #
  129. # article_list3 = []
  130. # for i in article_list3_ori:
  131. # title = i['title']
  132. # if hash_map.get(title):
  133. # continue
  134. # else:
  135. # article_list3.append(i)
  136. # hash_map[title] = 1
  137. # 第二步对article_list1, article_list3按照得分排序, 对article_list2按照播放量排序
  138. if article_list1_ori:
  139. rank1 = ArticleRank().rank(
  140. account_list=[self.accountName],
  141. text_list=[i['title'] for i in article_list1_ori]
  142. )
  143. score_list1 = rank1[self.accountName]['score_list']
  144. ranked_1 = []
  145. for index, value in enumerate(score_list1):
  146. obj = article_list1_ori[index]
  147. obj['score'] = value + 1000
  148. ranked_1.append(obj)
  149. ranked_1 = sorted(ranked_1, key=lambda x: x['score'], reverse=True)
  150. else:
  151. ranked_1 = []
  152. # rank2
  153. # if article_list2_ori:
  154. # for item in article_list2_ori:
  155. # item['score'] = 100
  156. # ranked_2 = sorted(article_list2_ori, key=lambda x: x['crawlerViewCount'], reverse=True)
  157. # else:
  158. ranked_2 = []
  159. # rank3
  160. if article_list3_ori:
  161. rank3 = ArticleRank().rank(
  162. account_list=[self.accountName],
  163. text_list=[i['title'] for i in article_list3_ori]
  164. )
  165. score_list3 = rank3[self.accountName]['score_list']
  166. ranked_3 = []
  167. for index, value in enumerate(score_list3):
  168. obj = article_list3_ori[index]
  169. obj['score'] = value
  170. ranked_3.append(obj)
  171. ranked_3 = sorted(ranked_3, key=lambda x: x['score'], reverse=True)
  172. else:
  173. ranked_3 = []
  174. self.logger.log(
  175. code="1004",
  176. msg="排序完成",
  177. data={
  178. "rank1": ranked_1,
  179. "rank2": ranked_2,
  180. "rank3": ranked_3
  181. }
  182. )
  183. return ranked_1, ranked_2, ranked_3
  184. async def rank_v1(self):
  185. """
  186. Rank Version 1
  187. :return:
  188. """
  189. print("开始排序")
  190. try:
  191. ranked_1_d, ranked_2_d, ranked_3_d = await self.basic_rank()
  192. ranked_1, ranked_2, ranked_3 = deduplication(ranked_1_d, ranked_2_d, ranked_3_d)
  193. print("去重成功")
  194. try:
  195. L = []
  196. if ranked_1:
  197. target = random.choice(ranked_1[:5])
  198. L.append(target)
  199. if ranked_2:
  200. L.append(ranked_2[0])
  201. else:
  202. if ranked_2:
  203. if len(ranked_2) > 1:
  204. for i in ranked_2[:2]:
  205. L.append(i)
  206. else:
  207. L.append(ranked_2[0])
  208. # L only 1
  209. for item in ranked_3:
  210. L.append(item)
  211. # L 1 and 3
  212. result = {
  213. "accountId": self.accountId,
  214. "accountName": self.accountName,
  215. "ghId": self.ghId,
  216. "strategy": self.strategy,
  217. "publishNum": self.publishNum,
  218. "rank_list": L[:self.publishNum],
  219. "filter_list": self.filter_list
  220. }
  221. self.logger.log(
  222. code=1006,
  223. msg="rank successfully",
  224. data=result
  225. )
  226. response = {"status": "Rank Success", "data": result, "code": 1}
  227. except Exception as e:
  228. result = {
  229. "accountId": self.accountId,
  230. "accountName": self.accountName,
  231. "ghId": self.ghId,
  232. "strategy": self.strategy,
  233. "publishNum": self.publishNum,
  234. "rank_list": self.publishArticleList[: self.publishNum],
  235. "filter_list": self.filter_list
  236. }
  237. self.logger.log(
  238. code=1007,
  239. msg="rank failed because of {}".format(e),
  240. data=result
  241. )
  242. print("排序成功")
  243. response = {"status": "Rank Fail", "data": result, "code": 1}
  244. return response
  245. except:
  246. result = {"code": 2, "info": "account is not exist"}
  247. return result
  248. async def rank_v2(self):
  249. """
  250. Rank Version 2
  251. :return:
  252. """
  253. return await self.rank_v1()
  254. async def rank_v3(self):
  255. """
  256. Rank Version 3
  257. :return:
  258. """
  259. return await self.rank_v1()
  260. async def rank_v4(self):
  261. """
  262. Rank Version 4
  263. :return:
  264. """
  265. return await self.rank_v1()
  266. async def rank_v5(self):
  267. """
  268. Rank Version 5
  269. :return:
  270. """
  271. return await self.rank_v1()
  272. async def choose_strategy(self):
  273. """
  274. 选择排序策略
  275. :return:
  276. """
  277. match self.strategy:
  278. case "ArticleRankV1":
  279. self.logger.log(
  280. code="1003",
  281. msg="命中排序策略1"
  282. )
  283. return await self.rank_v1()
  284. case "ArticleRankV2":
  285. self.logger.log(
  286. code="1003",
  287. msg="命中排序策略2"
  288. )
  289. return await self.rank_v2()
  290. case "ArticleRankV3":
  291. self.logger.log(
  292. code="1003",
  293. msg="命中排序策略3"
  294. )
  295. return await self.rank_v3()
  296. case "ArticleRankV4":
  297. self.logger.log(
  298. code="1003",
  299. msg="命中排序策略4"
  300. )
  301. return await self.rank_v4()
  302. case "ArticleRankV5":
  303. self.logger.log(
  304. code="1003",
  305. msg="命中排序策略5"
  306. )
  307. return await self.rank_v5()
  308. async def deal(self):
  309. """
  310. Deal Function
  311. :return:
  312. """
  313. error_params = await self.check_params()
  314. if error_params:
  315. return error_params
  316. else:
  317. print("参数校验成功")
  318. return await self.choose_strategy()