Commit 56367639 authored by yangruiqiang's avatar yangruiqiang

Merge remote-tracking branch 'origin/master'

parents 1447f261 1f4a799d
...@@ -1080,9 +1080,9 @@ if __name__ == "__main__": ...@@ -1080,9 +1080,9 @@ if __name__ == "__main__":
t1 = threading.Thread(target=scheduler_thread, daemon=True) t1 = threading.Thread(target=scheduler_thread, daemon=True)
t2 = threading.Thread(target=runner_thread, daemon=True) t2 = threading.Thread(target=runner_thread, daemon=True)
# t1.start() t1.start()
# t2.start() t2.start()
# t1.join() t1.join()
# t2.join() t2.join()
redis_client = init_redis4() # redis_client = init_redis4()
print(redis_client.delete('mt_third_task')) # print(redis_client.delete('mt_third_task'))
\ No newline at end of file \ No newline at end of file
...@@ -300,36 +300,31 @@ def doubao_mobile_process_original_data(data): ...@@ -300,36 +300,31 @@ def doubao_mobile_process_original_data(data):
url_list = new_url_list url_list = new_url_list
print(suggestions)
print(search_keyword)
print(response_content)
print(rich_media_block)
print(len(url_list))
spider_save_tos.process_and_save_files(file_path, search_keyword, url_list, think_content, response_content, spider_save_tos.process_and_save_files(file_path, search_keyword, url_list, think_content, response_content,
suggestions, rich_media_block) suggestions, rich_media_block)
return (file_path, search_keyword, url_list, think_content, response_content, suggestions) return (file_path, search_keyword, url_list, think_content, response_content, suggestions)
except Exception as e: except Exception as e:
traceback.print_exc() # traceback.print_exc()
# parts = file_path.split('/') parts = file_path.split('/')
# platform = parts[2] platform = parts[2]
# task_id = parts[1] task_id = parts[1]
# context, quote, suggestion, think, search_word = get_parse_sse_result(platform, task_id) context, quote, suggestion, think, search_word = get_parse_sse_result(platform, task_id)
# if context: if context:
# response_content = context response_content = context
# url_list = quote url_list = quote
# suggestions = suggestion suggestions = suggestion
# think_content = think think_content = think
# search_keyword = search_word search_keyword = search_word
# spider_save_tos.process_and_save_files_ai(file_path, search_keyword, url_list, think_content, spider_save_tos.process_and_save_files_ai(file_path, search_keyword, url_list, think_content,
# response_content,suggestions) response_content,suggestions)
#
# return (file_path, search_keyword, url_list, think_content, response_content, suggestions) return (file_path, search_keyword, url_list, think_content, response_content, suggestions)
# else: else:
# response_content = "对话信息获取失败:-200" response_content = "对话信息获取失败:-200"
# robot_utils.feishu_tobot(file_path) robot_utils.feishu_tobot(file_path)
# spider_save_tos.process_and_save_files(file_path, search_keyword, url_list, think_content, response_content, spider_save_tos.process_and_save_files(file_path, search_keyword, url_list, think_content, response_content,
# suggestions, rich_media_block) suggestions, rich_media_block)
# return (file_path, search_keyword, url_list, think_content, response_content, suggestions) return (file_path, search_keyword, url_list, think_content, response_content, suggestions)
if __name__ == '__main__': if __name__ == '__main__':
......
...@@ -1286,7 +1286,6 @@ def result_v2(response_content, data): ...@@ -1286,7 +1286,6 @@ def result_v2(response_content, data):
# ------------------------- # -------------------------
# 获取ai提及词 # 获取ai提及词
ai_word_list = cache_get_ai_brand_list(taskId, platform, response_content, prompt) ai_word_list = cache_get_ai_brand_list(taskId, platform, response_content, prompt)
print(ai_word_list)
# 获取所有词 # 获取所有词
all_word_list = [] all_word_list = []
if isinstance(ai_word_list, list): if isinstance(ai_word_list, list):
...@@ -1311,16 +1310,9 @@ def result_v2(response_content, data): ...@@ -1311,16 +1310,9 @@ def result_v2(response_content, data):
# 获取所有词的排名 word + rank_list # 获取所有词的排名 word + rank_list
all_word_rank_list = get_keyword_ranks(response_content, all_word_set_list) all_word_rank_list = get_keyword_ranks(response_content, all_word_set_list)
# 获取所有词的排名 word + rank + count # 获取所有词的排名 word + rank + count
print(all_word_rank_list)
print('----')
print('----')
print('----')
all_keyword_with_rank = convert_rank_data(all_word_rank_list) all_keyword_with_rank = convert_rank_data(all_word_rank_list)
print(all_keyword_with_rank)
print('---')
print('---')
print('---')
print('---')
# 获取所有词的品牌 # 获取所有词的品牌
all_keyword_with_brand = keyword_map_brand(all_word_list) all_keyword_with_brand = keyword_map_brand(all_word_list)
...@@ -1916,41 +1908,41 @@ def run_data(PAGE_SIZE,MAX_WORKERS): ...@@ -1916,41 +1908,41 @@ def run_data(PAGE_SIZE,MAX_WORKERS):
if __name__ == '__main__': if __name__ == '__main__':
data_list = bh_utils.query_data(f"select * from geo_commit_task where reqId = '84db7b12-88e7-4aaa-bc58-8d61f1051934'") # data_list = bh_utils.query_data(f"select * from geo_commit_task where reqId = '84db7b12-88e7-4aaa-bc58-8d61f1051934'")
# data_list = bh_utils.query_data(query_sql) # # data_list = bh_utils.query_data(query_sql)
# print(data_list) # # print(data_list)
# # # # # # #
# # # # # # #
# # # # # # # # # # #
def handle_item(i): # def handle_item(i):
if i.get('comWordsMap'): # if i.get('comWordsMap'):
i['comWordsMap'] = json.loads(i.get('comWordsMap')) # i['comWordsMap'] = json.loads(i.get('comWordsMap'))
if i.get('brandWords'): # if i.get('brandWords'):
i['brandWords'] = json.loads(i.get('brandWords')) # i['brandWords'] = json.loads(i.get('brandWords'))
if i.get('comWords'): # if i.get('comWords'):
i['comWords'] = json.loads(i.get('comWords')) # i['comWords'] = json.loads(i.get('comWords'))
if i.get('keywords'): # if i.get('keywords'):
i['keywords'] = json.loads(i.get('keywords')) # i['keywords'] = json.loads(i.get('keywords'))
if i.get('productWordsMap'): # if i.get('productWordsMap'):
i['productWordsMap'] = json.loads(i.get('productWordsMap')) # i['productWordsMap'] = json.loads(i.get('productWordsMap'))
type_t = i.get('type') # type_t = i.get('type')
# type_t = 'batch' # # type_t = 'batch'
#
# return task_send_queue(i,type_t) # # return task_send_queue(i,type_t)
return platform_process(i) # return platform_process(i)
#
if data_list: # if data_list:
with ThreadPoolExecutor(max_workers=30) as executor: # with ThreadPoolExecutor(max_workers=30) as executor:
futures = [executor.submit(handle_item, i) for i in data_list] # futures = [executor.submit(handle_item, i) for i in data_list]
#
for future in as_completed(futures): # for future in as_completed(futures):
try: # try:
future.result() # future.result()
except Exception as e: # except Exception as e:
logger.exception(f"platform_process 执行异常: {e}") # logger.exception(f"platform_process 执行异常: {e}")
# PAGE_SIZE =1000 PAGE_SIZE =1000
# MAX_WORKERS =50 MAX_WORKERS =50
# run_data(PAGE_SIZE,MAX_WORKERS) run_data(PAGE_SIZE,MAX_WORKERS)
text = """ text = """
""" """
......
...@@ -91,12 +91,7 @@ def qianwen_android_process_original_data(data): ...@@ -91,12 +91,7 @@ def qianwen_android_process_original_data(data):
if mime_type == 'multi_load/iframe' and multi_load_type == 'taoassistant_fold_product_feeds' and status == 'complete': if mime_type == 'multi_load/iframe' and multi_load_type == 'taoassistant_fold_product_feeds' and status == 'complete':
for mu in multi_load: for mu in multi_load:
print(mu)
print('---')
print('---')
print('---')
print('---')
print('---')
if mu.get('type') == 'taoassistant_fold_product_feeds': if mu.get('type') == 'taoassistant_fold_product_feeds':
source_seq = mu.get('source_seq') source_seq = mu.get('source_seq')
jump_url = next( jump_url = next(
...@@ -213,7 +208,6 @@ def qianwen_android_process_original_data(data): ...@@ -213,7 +208,6 @@ def qianwen_android_process_original_data(data):
if url_list_batch: if url_list_batch:
url_list = url_list_batch url_list = url_list_batch
print(response_content)
spider_save_tos.process_and_save_files(file_path, search_keyword, url_list, think_content, response_content, spider_save_tos.process_and_save_files(file_path, search_keyword, url_list, think_content, response_content,
suggestions, rich_media_block) suggestions, rich_media_block)
return (file_path, search_keyword, url_list, think_content, response_content, suggestions) return (file_path, search_keyword, url_list, think_content, response_content, suggestions)
......
Markdown is supported
0% or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment