accountArticleRank.py 12 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357
  1. """
  2. @author: luojunhui
  3. """
  4. import random
  5. import time
  6. from uuid import uuid4
  7. from tqdm import tqdm
  8. from applications.aliyunLog import AliyunArticleLog
  9. from applications.functions import ArticleRank, title_sim_v2_by_list
  10. from applications.pipeline import LongArticlesPipeline
  11. def deduplication(rank1, rank2, rank3):
  12. """
  13. 标题相似度去重
  14. :return:
  15. """
  16. dup_list = []
  17. final_result = []
  18. for item_list in [rank1, rank2, rank3]:
  19. result = []
  20. if item_list:
  21. for item in item_list:
  22. title = item["title"]
  23. if title_sim_v2_by_list(title, dup_list):
  24. # print("标题重复,已经过滤\t", title)
  25. continue
  26. else:
  27. result.append(item)
  28. dup_list.append(title)
  29. final_result.append(result)
  30. return final_result[0], final_result[1], final_result[2]
  31. class AccountArticleRank(object):
  32. """
  33. 文章排序
  34. """
  35. def __init__(self, params, mysql_client):
  36. """
  37. :param params: 请求参数
  38. :param mysql_client: 数据库链接池
  39. """
  40. self.filter_list = None
  41. self.publishArticleList = None
  42. self.publishNum = None
  43. self.strategy = None
  44. self.ghId = None
  45. self.accountName = None
  46. self.accountId = None
  47. self.params = params
  48. self.mysql_client = mysql_client
  49. self.request_id = "alg-{}-{}".format(uuid4(), int(time.time()))
  50. self.logger = AliyunArticleLog(request_id=self.request_id, alg="ArticleRank")
  51. self.pipeline = LongArticlesPipeline()
  52. def filter(self):
  53. """
  54. 过滤器
  55. """
  56. self.publishArticleList = []
  57. self.filter_list = []
  58. print("历史")
  59. history_title_dict = self.pipeline.history_title(
  60. account_nickname=self.accountName
  61. )
  62. print(history_title_dict)
  63. for item in tqdm(self.params["publishArticleList"]):
  64. flag = self.pipeline.deal(item, self.accountName, history_title_dict)
  65. if flag:
  66. item["filterReason"] = flag["filterReason"]
  67. self.filter_list.append(item)
  68. else:
  69. self.publishArticleList.append(item)
  70. print("过滤完成")
  71. async def check_params(self):
  72. """
  73. 校验参数
  74. :return:
  75. """
  76. try:
  77. self.accountId = self.params["accountId"]
  78. self.accountName = self.params["accountName"]
  79. self.ghId = self.params["ghId"]
  80. self.strategy = self.params["strategy"]
  81. self.publishNum = self.params["publishNum"]
  82. print("开始校验参数")
  83. self.filter()
  84. print("参数校验成功")
  85. self.logger.log(code="1001", msg="参数校验成功", data=self.params)
  86. return None
  87. except Exception as e:
  88. response = {
  89. "msg": "params error",
  90. "info": "params check failed, params : {} is not correct".format(e),
  91. "code": 0,
  92. }
  93. self.logger.log(
  94. code="1002", msg="参数校验失败--{}".format(e), data=self.params
  95. )
  96. return response
  97. async def basic_rank(self):
  98. """
  99. 基础排序
  100. :return:
  101. """
  102. # 第一步把所有文章标题分为3组
  103. article_list1_ori = [
  104. i for i in self.publishArticleList if "【1】" in i["producePlanName"]
  105. ]
  106. article_list2_ori = [
  107. i for i in self.publishArticleList if "【2】" in i["producePlanName"]
  108. ]
  109. article_list3_ori = [
  110. i
  111. for i in self.publishArticleList
  112. if not i in article_list1_ori and not i in article_list2_ori
  113. ]
  114. # # 全局去重,保留优先级由 L1 --> L2 --> L3
  115. # hash_map = {}
  116. #
  117. # article_list1 = []
  118. # for i in article_list1_ori:
  119. # title = i['title']
  120. # if hash_map.get(title):
  121. # continue
  122. # else:
  123. # article_list1.append(i)
  124. # hash_map[title] = 1
  125. #
  126. # article_list2 = []
  127. # for i in article_list2_ori:
  128. # title = i['title']
  129. # if hash_map.get(title):
  130. # continue
  131. # else:
  132. # article_list2.append(i)
  133. # hash_map[title] = 2
  134. #
  135. # article_list3 = []
  136. # for i in article_list3_ori:
  137. # title = i['title']
  138. # if hash_map.get(title):
  139. # continue
  140. # else:
  141. # article_list3.append(i)
  142. # hash_map[title] = 1
  143. # 第二步对article_list1, article_list3按照得分排序, 对article_list2按照播放量排序
  144. if article_list1_ori:
  145. rank1 = ArticleRank().rank(
  146. account_list=[self.accountName],
  147. text_list=[i["title"] for i in article_list1_ori],
  148. )
  149. score_list1 = rank1[self.accountName]["score_list"]
  150. ranked_1 = []
  151. for index, value in enumerate(score_list1):
  152. obj = article_list1_ori[index]
  153. obj["score"] = value + 1000
  154. ranked_1.append(obj)
  155. ranked_1 = sorted(ranked_1, key=lambda x: x["score"], reverse=True)
  156. else:
  157. ranked_1 = []
  158. # rank2
  159. if article_list2_ori:
  160. for item in article_list2_ori:
  161. item["score"] = 100
  162. ranked_2 = sorted(
  163. article_list2_ori, key=lambda x: x["crawlerViewCount"], reverse=True
  164. )
  165. else:
  166. ranked_2 = []
  167. # rank3
  168. if article_list3_ori:
  169. rank3 = ArticleRank().rank(
  170. account_list=[self.accountName],
  171. text_list=[i["title"] for i in article_list3_ori],
  172. )
  173. score_list3 = rank3[self.accountName]["score_list"]
  174. ranked_3 = []
  175. for index, value in enumerate(score_list3):
  176. obj = article_list3_ori[index]
  177. obj["score"] = value
  178. ranked_3.append(obj)
  179. ranked_3 = sorted(ranked_3, key=lambda x: x["score"], reverse=True)
  180. else:
  181. ranked_3 = []
  182. self.logger.log(
  183. code="1004",
  184. msg="排序完成",
  185. data={"rank1": ranked_1, "rank2": ranked_2, "rank3": ranked_3},
  186. )
  187. return ranked_1, ranked_2, ranked_3
  188. async def rank_v1(self):
  189. """
  190. Rank Version 1
  191. :return:
  192. """
  193. print("开始排序")
  194. try:
  195. ranked_1_d, ranked_2_d, ranked_3_d = await self.basic_rank()
  196. ranked_1, ranked_2, ranked_3 = deduplication(
  197. ranked_1_d, ranked_2_d, ranked_3_d
  198. )
  199. print("去重成功")
  200. try:
  201. L = []
  202. if ranked_1:
  203. target = random.choice(ranked_1[:5])
  204. L.append(target)
  205. if ranked_2:
  206. L.append(ranked_2[0])
  207. else:
  208. if ranked_2:
  209. if len(ranked_2) > 1:
  210. for i in ranked_2[:2]:
  211. L.append(i)
  212. else:
  213. L.append(ranked_2[0])
  214. # L only 1
  215. for item in ranked_3:
  216. L.append(item)
  217. # L 1 and 3
  218. result = {
  219. "accountId": self.accountId,
  220. "accountName": self.accountName,
  221. "ghId": self.ghId,
  222. "strategy": self.strategy,
  223. "publishNum": self.publishNum,
  224. "rank_list": L[: self.publishNum],
  225. "filter_list": self.filter_list,
  226. }
  227. self.logger.log(code=1006, msg="rank successfully", data=result)
  228. response = {"status": "Rank Success", "data": result, "code": 1}
  229. except Exception as e:
  230. result = {
  231. "accountId": self.accountId,
  232. "accountName": self.accountName,
  233. "ghId": self.ghId,
  234. "strategy": self.strategy,
  235. "publishNum": self.publishNum,
  236. "rank_list": self.publishArticleList[: self.publishNum],
  237. "filter_list": self.filter_list,
  238. }
  239. self.logger.log(
  240. code=1007, msg="rank failed because of {}".format(e), data=result
  241. )
  242. print("排序成功")
  243. response = {"status": "Rank Fail", "data": result, "code": 1}
  244. return response
  245. except:
  246. result = {"code": 2, "info": "account is not exist"}
  247. return result
  248. async def rank_v2(self):
  249. """
  250. Rank Version 2
  251. :return:
  252. """
  253. try:
  254. ranks = ArticleRank().rank(
  255. account_list=[self.accountName],
  256. text_list=[i["title"] for i in self.publishArticleList],
  257. )
  258. score_list1 = ranks[self.accountName]["score_list"]
  259. ranked_v2 = []
  260. for index, value in enumerate(score_list1):
  261. obj = self.publishArticleList[index]
  262. obj["score"] = value
  263. ranked_v2.append(obj)
  264. ranked_v2 = sorted(ranked_v2, key=lambda x: (-x["score"], -x['crawlerViewCount']))
  265. result = {
  266. "accountId": self.accountId,
  267. "accountName": self.accountName,
  268. "ghId": self.ghId,
  269. "strategy": self.strategy,
  270. "publishNum": self.publishNum,
  271. "rank_list": ranked_v2[: self.publishNum],
  272. "filter_list": self.filter_list,
  273. }
  274. response = {"status": "Rank Success", "data": result, "code": 1}
  275. except Exception as e:
  276. result = {
  277. "accountId": self.accountId,
  278. "accountName": self.accountName,
  279. "ghId": self.ghId,
  280. "strategy": self.strategy,
  281. "publishNum": self.publishNum,
  282. "rank_list": self.publishArticleList[: self.publishNum],
  283. "filter_list": self.filter_list,
  284. }
  285. response = {"status": "Rank Fail Because Of {}".format(e), "data": result, "code": 1}
  286. return response
  287. async def rank_v3(self):
  288. """
  289. Rank Version 3
  290. :return:
  291. """
  292. return await self.rank_v1()
  293. async def rank_v4(self):
  294. """
  295. Rank Version 4
  296. :return:
  297. """
  298. return await self.rank_v1()
  299. async def rank_v5(self):
  300. """
  301. Rank Version 5
  302. :return:
  303. """
  304. return await self.rank_v1()
  305. async def choose_strategy(self):
  306. """
  307. 选择排序策略
  308. :return:
  309. """
  310. match self.strategy:
  311. case "ArticleRankV1":
  312. self.logger.log(code="1003", msg="命中排序策略1")
  313. return await self.rank_v1()
  314. case "ArticleRankV2":
  315. self.logger.log(code="1003", msg="命中排序策略2")
  316. return await self.rank_v2()
  317. case "ArticleRankV3":
  318. self.logger.log(code="1003", msg="命中排序策略3")
  319. return await self.rank_v3()
  320. case "ArticleRankV4":
  321. self.logger.log(code="1003", msg="命中排序策略4")
  322. return await self.rank_v4()
  323. case "ArticleRankV5":
  324. self.logger.log(code="1003", msg="命中排序策略5")
  325. return await self.rank_v5()
  326. async def deal(self):
  327. """
  328. Deal Function
  329. :return:
  330. """
  331. error_params = await self.check_params()
  332. if error_params:
  333. return error_params
  334. else:
  335. print("参数校验成功")
  336. return await self.choose_strategy()