diff --git a/.gitignore b/.gitignore index 185e663..7d699f5 100644 --- a/.gitignore +++ b/.gitignore @@ -11,6 +11,8 @@ npm-debug.log* yarn-debug.log* yarn-error.log* +#conf +backend/secrets.py # Editor directories and files .idea .vscode diff --git a/_data/foo.py b/_data/foo.py index d6e6093..8d3b0b4 100644 --- a/_data/foo.py +++ b/_data/foo.py @@ -1,6 +1,6 @@ #!/usr/bin/env python # -*- coding: utf-8 -*- -# Created by Administrator at 2020/4/14 23:18 +# Created by imoyao at 2020/4/14 23:18 # // f for fontColor / d for dark / b for bright # // c for color series # // r red ;b black ;w white ;p purple ;c cyan ;g green ;y yellow ; diff --git a/_data/lipsticks/999_metal.jpg b/_data/lipsticks/999_metal.jpg new file mode 100644 index 0000000..397f52d Binary files /dev/null and b/_data/lipsticks/999_metal.jpg differ diff --git a/_data/lipsticks/999_zirun.jpg b/_data/lipsticks/999_zirun.jpg new file mode 100644 index 0000000..4e967e9 Binary files /dev/null and b/_data/lipsticks/999_zirun.jpg differ diff --git a/_data/lipsticks/Dior/lylj/127266.jpg b/_data/lipsticks/Dior/lylj/127266.jpg new file mode 100644 index 0000000..f3d93a6 Binary files /dev/null and b/_data/lipsticks/Dior/lylj/127266.jpg differ diff --git a/_data/lipsticks/Dior/lylj/127267.jpg b/_data/lipsticks/Dior/lylj/127267.jpg new file mode 100644 index 0000000..98838a4 Binary files /dev/null and b/_data/lipsticks/Dior/lylj/127267.jpg differ diff --git a/_data/lipsticks/Dior/lylj/127268.jpg b/_data/lipsticks/Dior/lylj/127268.jpg new file mode 100644 index 0000000..3c7009c Binary files /dev/null and b/_data/lipsticks/Dior/lylj/127268.jpg differ diff --git a/_data/lipsticks/Dior/lylj/127269.jpg b/_data/lipsticks/Dior/lylj/127269.jpg new file mode 100644 index 0000000..cbdbbf0 Binary files /dev/null and b/_data/lipsticks/Dior/lylj/127269.jpg differ diff --git a/_data/lipsticks/Dior/lylj/127270.jpg b/_data/lipsticks/Dior/lylj/127270.jpg new file mode 100644 index 0000000..397f52d Binary files /dev/null and b/_data/lipsticks/Dior/lylj/127270.jpg differ diff --git a/_data/lipsticks/Dior/lylj/127271.jpg b/_data/lipsticks/Dior/lylj/127271.jpg new file mode 100644 index 0000000..4e967e9 Binary files /dev/null and b/_data/lipsticks/Dior/lylj/127271.jpg differ diff --git a/_data/lipsticks/Dior/lylj/127272.jpg b/_data/lipsticks/Dior/lylj/127272.jpg new file mode 100644 index 0000000..93c7085 Binary files /dev/null and b/_data/lipsticks/Dior/lylj/127272.jpg differ diff --git a/_data/lipsticks/big-jpg.png b/_data/lipsticks/big-jpg.png new file mode 100644 index 0000000..f1ab2d1 Binary files /dev/null and b/_data/lipsticks/big-jpg.png differ diff --git a/_data/lipsticks/many.png b/_data/lipsticks/many.png new file mode 100644 index 0000000..a5d7eec Binary files /dev/null and b/_data/lipsticks/many.png differ diff --git a/_data/src/mp.png b/_data/src/mp.png new file mode 100644 index 0000000..b6c8b5a Binary files /dev/null and b/_data/src/mp.png differ diff --git a/_data/src/mp_test.png b/_data/src/mp_test.png new file mode 100644 index 0000000..b49a787 Binary files /dev/null and b/_data/src/mp_test.png differ diff --git a/_data/src/test.png b/_data/src/test.png new file mode 100644 index 0000000..92b6e88 Binary files /dev/null and b/_data/src/test.png differ diff --git a/_data/src/test_data.png b/_data/src/test_data.png new file mode 100644 index 0000000..974e71b Binary files /dev/null and b/_data/src/test_data.png differ diff --git a/_data/xiji/products.html b/_data/xiji/products.html new file mode 100644 index 0000000..9797477 --- /dev/null +++ b/_data/xiji/products.html @@ -0,0 +1,7522 @@ + + + + + +Dior 迪奥 烈艳蓝金唇膏/口红 3.5g 多色可选【价格 、图片、评价】- 西集网 + + + + + + + + +
+ + +
+
+
+ 返回首页
+ 配送至: +
+
+
+
+ + + + +
+
+ + +
+
+
+ + + + + + + + + +
+
+ +
+ +
+
+ + + +
+
+ +
+ + + + + + + +
+ +
+ +
+
+ Dior 迪奥 烈艳蓝金唇膏/口红 3.5g 多色可选 +
+ + + + +
+
+ +
+
    +
  • +
    +
    Dior 迪奥 烈艳蓝金唇膏/口红 3.5g 多色可选
    +
  • +
+
+ +
+
+ +
+ +
+
    +
  • + 销量331 +
  • + +
  • + 浏览1.1W +
  • +
  • + + +
    +
    +
    告诉微信小伙伴去
    +
    + +
    + +
  • +
+
+ + +
+ +
+ + +
+ +
+
+
+ Dior/迪奥 +
+ +
+ 自营 + 香港直邮 +
+
+ +
+
+

+ Dior 迪奥 烈艳蓝金唇膏/口红 3.5g 多色可选  #520

+
+
+
+
+
+ 迪奥烈艳蓝金唇膏宣告了护唇与上妆相冲突的时代已然结束,双唇既能闪耀魅力色彩,同时又能备受滋养呵护。颜色很正,而且质地丝滑细腻,涂抹后可以让双唇更加饱满,让你爱不释手。 +
+
+
+
+ + + + + + + + + + + + +
+
    +
  • +
    + + +
    + + 特卖 + + 距离结束还有 + 00天 + 00时 + 00分 + 00秒 + 00 + +
    +
    + + 降价通知 + 售价 + + + + + + + + + + + +
    +
    +
    税费
    本商品由商家补贴税费税费收取规则
    +
    +
    + +
    + +
  • + +
  • 此海外商品售价会随实时汇率波动产生变化,以您订单生成显示的金额为准。

  • + + + + + + + + + + + + + + + + +
+ + + +
+ + + + + + + + + + + + + +
+ +
+ +
+ + + + + + + + +
+
+ + +
+
    +
  • 重量
  • +
+
+ + +
+ + +
+ +
    + +
  • + + + -+ + + + + +
  • + +
  • 请注意 国家药监局提示您:请正确认识化妆品功效,化妆品不能替代药品,不能治疗皮肤病等疾病。由于拍摄光线等问题,可能存在色差,不同批次,厂家可能更改包装,产品以收到实物为准!

  • +
+ + + +
+
+ +
+ +
+ + + + + + + + + +
+ +
+ +
+
+ + +
+
+ + + +
+
+ +
+ + + + + + + +
+ +
+ + +
+ + + +
+ +
+
+
    +
+
+ +
+ + + +
+ +
+ + + + +
+
+ + + + + + + + +
+
+

商品详情

+ + +
+
+ +

商品详情

+ +
+ + +
+
+ + + + + + + + + + + + + + + + + + +
+ + +
+
+
+
+ + + +
+ + +
+
    +
+
+ + +
+
+ +
+
+ + +
+
+ + + + + + + + + + + + + + + + + + + + +
+
+ +
+ + + + + + + + + +
+ +
+ + + + + + + + + diff --git a/backend/core/color_parse.py b/backend/core/color_parse.py index 4f61fca..c7fec04 100644 --- a/backend/core/color_parse.py +++ b/backend/core/color_parse.py @@ -9,14 +9,13 @@ import re from collections import defaultdict from functools import wraps, partial -import itertools import json import yaml from pypinyin import lazy_pinyin, Style from backend.libs.zhtools.langconv import Converter -from backend import settings +from backend import settings, utils class ConvertColor: @@ -237,8 +236,8 @@ def parse_lipstick(self): brand_list = [] for brand in lipstick_data.get('brands'): series_list = [] - brand_zh_name = brand.get('name','') - brand_en_name = brand.get('en_name','') + brand_zh_name = brand.get('name', '') + brand_en_name = brand.get('en_name', '') if not brand_en_name: brand_en_name = '_'.join(lazy_pinyin(brand_zh_name)) series_name = brand.get('series') @@ -298,7 +297,7 @@ def parse_nippon_color(self, group_data=False, dump_data=False, group_by='color_ color_obj = self.color_object_maker(name, color_hex, color_rgb=color_rgb, pinyin_str=jp_pinyin_str, color_cmyk=color_cmyk, is_simple=False) jp_list.append(color_obj) - all_in_one = merge_iterables_of_dict('id', nippor_list, jp_list) + all_in_one = utils.merge_iterables_of_dict('id', nippor_list, jp_list) if dump_data: grouped_data = [] if group_data: # 此处只在导出前分组,没有对各组数据分别分组 @@ -442,7 +441,7 @@ def all_in_one(self, setting_obj, group_data=False, dump_data=False, group_by='c colors_data = self.parse_flinhong(settings.FLINHONG_COLORS_INFO, dump_data=dump_data) cfs_color_data = self.parse_cfs_color(settings.CFS_COLOR_INFO, dump_data=dump_data) # chinese_colors_data 放后面,因为有描述和图片 - all_in_one = merge_iterables_of_dict('id', jizhi_data, colors_data, cfs_color_data, chinese_colors_data) + all_in_one = utils.merge_iterables_of_dict('id', jizhi_data, colors_data, cfs_color_data, chinese_colors_data) # print(type(all_in_one), all_in_one) print('before_filter:', len(jizhi_data) + len(chinese_colors_data) + len(colors_data) + len(cfs_color_data)) print('after_filter:', len(all_in_one)) @@ -491,24 +490,6 @@ def group_iterables_of_dicts_in_list(group_key, iterables): return row_by_key -def merge_iterables_of_dict(shared_key, *iterables): - """ - see also:[🐍PyTricks | Python 中如何合并一个内字典列表? | 别院牧志](https://imoyao.github.io/blog/2020-04-19/python-merge-two-list-of-dicts/) - chinese_colors_data 放前面,因为有描述和图片 - :param shared_key: - :param iterables: - :return: - """ - result = defaultdict(dict) - for dictionary in itertools.chain.from_iterable(iterables): - result[dictionary[shared_key]].update(dictionary) - # for dictionary in result.values(): - # dictionary.pop(shared_key) - # return result - result = list(result.values()) # 保证返回为list,否则:TypeError: Object of type dict_values is not JSON serializable - return result - - def update_by_value(v): """ 根据 V 值去更新色系数据 @@ -570,7 +551,7 @@ def set_color_name(new_name): converter = ConvertColor() -@find_color_series_by_name(name='') +# @find_color_series_by_name(name='') def find_color_series(rgb_seq): # TODO:此处是否有更好实现?cmyk去判断是否是100%cmy颜色(黑色不判断) """ TODO: see also: https://github.com/MisanthropicBit/colorise/blob/master/colorise/color_tools.py @@ -659,14 +640,7 @@ def unify_color_dict(color): if __name__ == '__main__': - color_list = settings.COLOR_BASE_MAP.values() + color_list = [[22, 24, 35], [36, 134, 185], [234, 137, 88], [32, 161, 98], [100, 106, 88]] for item in color_list: print(find_color_series(item)) - # import colorsys - # print(colorsys.rgb_to_hsv(*item)) - # print('rgb_to_hsv:', rgb_to_hsv(item)) - # print('rgb_to_hsv_org:', rgb_to_hsv_org(item)) - print('------------------') - a = [100, 106, 88] - print(find_color_series(a)) diff --git a/backend/core/img_ocr.py b/backend/core/img_ocr.py new file mode 100644 index 0000000..bd5f3ec --- /dev/null +++ b/backend/core/img_ocr.py @@ -0,0 +1,106 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# Created by imoyao at 2020/5/1 22:49 +import time +from functools import wraps + +from PIL import Image +from backend.libs.baidu_api.aip import AipOcr + +from backend import secrets, settings + + +class BaiduOCR: + def __init__(self): + self.client = AipOcr(secrets.APP_ID, secrets.API_KEY, secrets.SECRET_KEY) + self.options = {"language_type": "CHN_ENG", "detect_direction": "true", "detect_language": "true", + "probability": "true"} + + @staticmethod + def get_file_content(file_path): + """ + 读取图片 + """ + with open(file_path, 'rb') as fp: + return fp.read() + + def basic_parse(self, fp, set_option=False): + image = self.get_file_content(fp) + option = self.options if set_option else None + ret = self.client.basicGeneral(image, options=option) + return ret + + def basic_accurate(self, fp, set_option=False): + image = self.get_file_content(fp) + option = self.options if set_option else None + ret = self.client.basicAccurate(image, options=option) + return ret + + def basic_parse_url(self, url, set_option=False): + option = self.options if set_option else None + ret = self.client.basicGeneralUrl(url, options=option) + return ret + + def main(self, fp): + ret_data = self.basic_parse(fp) + return ret_data + + +baidu_ocr = BaiduOCR() + + +def time_it(func): + @wraps(func) + def wrapper(*args, **kwargs): + start_time = time.time() + res = func(*args, **kwargs) + end_time = time.time() + cost_time = end_time - start_time + return func.__name__, res, cost_time + + return wrapper + + +# @time_it +def get_rgb_of_img_getpixel(fp): + """ + 此处说这个方法比下面的慢,测试相反,需要进一步验证:https://www.cnblogs.com/chimeiwangliang/p/7130434.html + :param fp: + :return: + """ + img = Image.open(fp) + rgb_color = img.getpixel((96, 720)) + return rgb_color + + +# @time_it +def get_rgb_of_img_load(fp): + im = Image.open(fp) # Can be many different formats. + pix = im.load() + # return im.size # Get the width and hight of the image for iterating over + return pix[96, 720] # Get the RGBA Value of the a pixel of an image + # pix[x, y] = value # Set the RGBA Value of the image (tuple) + # im.save('alive_parrot.png') # Save the modified pixels as .png + + +if __name__ == '__main__': + # fp = '../../_data/lipsticks/many.png' + fp = '../../_data/lipsticks/big-jpg.png' + ''':param + + 1: {'log_id': 997397551332359778, 'direction': 0, 'words_result_num': 24, 'words_result': [{'words': '01排', 'probability': {'variance': 0.023368, 'average': 0.853212, 'min': 0.640628}}, {'words': '04排', 'probability': {'variance': 0.062432, 'average': 0.81162, 'min': 0.45852}}, {'words': '08', 'probability': {'variance': 0.002019, 'average': 0.946064, 'min': 0.901128}}, {'words': '09', 'probability': {'variance': 0.000805, 'average': 0.951882, 'min': 0.923516}}, {'words': '16', 'probability': {'variance': 0.0, 'average': 0.999542, 'min': 0.999384}}, {'words': 'AN P KAN ROSE FLAMNGO TRUE CORAL SCARLET ROUOE', 'probability': {'variance': 0.047922, 'average': 0.617961, 'min': 0.16106}}, {'words': '10排', 'probability': {'variance': 0.061092, 'average': 0.824526, 'min': 0.474978}}, {'words': '21', 'probability': {'variance': 0.000295, 'average': 0.980994, 'min': 0.963814}}, {'words': '22#', 'probability': {'variance': 3.7e-05, 'average': 0.993604, 'min': 0.985051}}, {'words': '13排', 'probability': {'variance': 0.054925, 'average': 0.815988, 'min': 0.485028}}, {'words': 'CHERRY LUSH VOLET FATALE NA D CORAL', 'probability': {'variance': 0.025493, 'average': 0.680056, 'min': 0.478117}}, {'words': '5', 'probability': {'variance': 0.0, 'average': 0.994346, 'min': 0.994346}}, {'words': '47', 'probability': {'variance': 0.0, 'average': 0.99995, 'min': 0.999927}}, {'words': '49', 'probability': {'variance': 4e-06, 'average': 0.997941, 'min': 0.99594}}, {'words': '15排', 'probability': {'variance': 0.000654, 'average': 0.980801, 'min': 0.944664}}, {'words': '23排', 'probability': {'variance': 0.042638, 'average': 0.852769, 'min': 0.560751}}, {'words': 'SHOWGIRL LLAC NYMPH MSEMAVED WLD ONER BARE PEACH', 'probability': {'variance': 0.019127, 'average': 0.637065, 'min': 0.357038}}, {'words': '35', 'probability': {'variance': 0.0, 'average': 0.999222, 'min': 0.998675}}, {'words': '14排', 'probability': {'variance': 0.002354, 'average': 0.962852, 'min': 0.894295}}, {'words': '7排', 'probability': {'variance': 0.043623, 'average': 0.697632, 'min': 0.488772}}, {'words': '03排', 'probability': {'variance': 0.063906, 'average': 0.812009, 'min': 0.454596}}, {'words': '46', 'probability': {'variance': 0.0, 'average': 0.999721, 'min': 0.999466}}, {'words': 'MSTE SAELE SMOKE', 'probability': {'variance': 0.025206, 'average': 0.776894, 'min': 0.553879}}, {'words': 'NK DUSK CASABLANCE SOMETHNOWLD', 'probability': {'variance': 0.017596, 'average': 0.83199, 'min': 0.609927}}], 'language': -1} + + 2: {'log_id': 7659937361977788002, 'words_result_num': 24, 'words_result': [{'words': '01#'}, {'words': '04#'}, {'words': '08#'}, {'words': '09#'}, {'words': '16#'}, {'words': 'SPANISH PO NDWN ROSE FLAMINGO TRUE CORAL SCARLET ROUC'}, {'words': '10#'}, {'words': '17#'}, {'words': '21#'}, {'words': '22#'}, {'words': '13#'}, {'words': 'CHERRY LUSH MOLET FATALE NOED CORNL. DOON PNX BLUSH NUDE'}, {'words': '45#'}, {'words': '47#'}, {'words': '49#'}, {'words': '15#'}, {'words': '23#'}, {'words': 'SHOWGIRLLLAC NYMPH MSBEHAVEDWLD ONCER BARE PEACH'}, {'words': '35#'}, {'words': '14#'}, {'words': '7#'}, {'words': '03#'}, {'words': '46#'}, {'words': 'SMELT MSTBRYSABLE SMOKE PINK DUSK CASABLANCESOMETHNOVLD'}]} + + 3:{'log_id': 3261191746421964930, 'words_result_num': 26, 'words_result': [{'words': '01#'}, {'words': '04#'}, {'words': '08#'}, {'words': '09#'}, {'words': '16#'}, {'words': ' SPANISH PI DNOUN ROSE FLAMINGO TRUECORAL SCARLET ROUO'}, {'words': '10#'}, {'words': '17#'}, {'words': '21#'}, {'words': '22#'}, {'words': '13#'}, {'words': ' CHERRY LUSH VOLET FATALE NACED CORAL FORSDGEN BLUSHNUDE'}, {'words': '45#'}, {'words': '47#'}, {'words': '49#'}, {'words': '15#'}, {'words': '23#'}, {'words': ' SHOWGIRL ULAC NYMPH MSH8HVED WLD ONGER BARE PEACH'}, {'words': '35#'}, {'words': '14#'}, {'words': '7#'}, {'words': '03#'}, {'words': '46#'}, {'words': ' SWCETMYSTERY SABLE SMOKE'}, {'words': ' PINKDUSK'}, {'words': ' CASABLANCE SOMETNGWLD'}]} + + 4:{'log_id': 4496045203345301378, 'words_result_num': 24, 'words_result': [{'words': '01#'}, {'words': '04#'}, {'words': '08#'}, {'words': '09#'}, {'words': '16#'}, {'words': ' SPAISH NDUN ROSE FLAMINGO TRUE CORAL SCARLET ROUCE'}, {'words': '10#'}, {'words': '17#'}, {'words': '21#'}, {'words': '22#'}, {'words': '13#'}, {'words': ' CHERRY LUSHVOLET FATALE NNCED CORAL FORSDOON BLUSH NUDE'}, {'words': '45#'}, {'words': '47#'}, {'words': '49#'}, {'words': '15#'}, {'words': '23#'}, {'words': ' SHOWGIRL UAC NYMPH MSBEHAV WLD ONNOER BARE PEACH'}, {'words': '35#'}, {'words': '14#'}, {'words': '7#'}, {'words': '03#'}, {'words': '46#'}, {'words': ' TMYSIYSABLE SMOKE PINK DUSK CASABUANCE SOMCTENGVOLD'}]} + ''' + bd_ocr = BaiduOCR() + ret = bd_ocr.basic_accurate(fp) + print(ret) + # for file in settings.TEST_IMAGE_FP: + # # print(file) + # colors = get_rgb_of_img_load(file) + # ano_color = get_rgb_of_img_getpixel(file) + # print(colors, ano_color) diff --git a/backend/core/xiji_parse.py b/backend/core/xiji_parse.py new file mode 100644 index 0000000..d2a5a5e --- /dev/null +++ b/backend/core/xiji_parse.py @@ -0,0 +1,102 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# Created by Administrator at 2020/5/8 23:24 +import os +import re + +from backend.core import img_ocr, xiji_spider +from backend import settings, utils +# from backend.core.img_ocr import time_it + +lylj = xiji_spider.LYLJ() + + +# @time_it +def get_item_detail(dir_p): + """ + 1. 获取图片rgb值 + 2. 对数据进行ocr识别,获取描述 + 3. 对识别结果进行解析处理 + 4. 组装数据 + :param dir_p: + :return: + """ + all_li = [] + for root, dirs, files in os.walk(dir_p): + for f in files: + fp = os.path.join(root, f) + fn = os.path.basename(fp) + only_name = fn.split('.')[0] + rgb = img_ocr.get_rgb_of_img_load(fp) # 1 + ocr_text = img_ocr.baidu_ocr.basic_accurate(fp) # 2 + + words_result = ocr_text.get('words_result') # 3 + ocr_obj = {} + for index, word in enumerate(words_result): + if '(' in word.get('words'): + subtitle = words_result[index].get('words') + re_ret = re.match(settings.REG_LYLJ_SUBTITLE_EXP, subtitle) + if re_ret: + real_subtitle = re_ret.group(1) + ocr_obj['subtile'] = real_subtitle + desc_list = words_result[index + 1:] + desc = ''.join([desc.get('words') for desc in desc_list]) + ocr_obj['desc'] = desc + + lipsticks_obj = { # 4 + 'rgb': rgb, + 'id': only_name, + } + lipsticks_obj.update(ocr_obj) + all_li.append(lipsticks_obj) + return all_li + + +# @time_it +def get_lylj_series(): + """ + 1. 去网络抓取数据 + 2. 获取抓取数据的信息 + 3. 以id为基准进行合并 + + [{'id': '127266', 'name': '#520', + 'src': 'https://img0.xiji.com/images/19/01/5b96671ae769403827ec84b8200805abb36a40a0.jpg?1577773530#w', + 'rgb': (247, 0, 83), 'subtile': '爱情水红恋爱中的粉红', 'desc': '这是一支非常有寓意的口红,520我爱你,表白专属色。这是散发着恋爱中粉红泡泡的颜色,暧昧、热恋,都洋溢在唇间。'}, + {'id': '127267', 'name': '#080', + 'src': 'https://img3.xiji.com/images/19/01/cf3cc958da0bc2b3715b69038ad417ef29523110.jpg?1577773529#w', + 'rgb': (220, 2, 3), 'subtile': '微笑正红春晚同款色', 'desc': '这款也是正红偏橘的色调,红多橘少,像是血橙的颜色,清新诱人,更适合日常使用。滋润质地,对唇部非常友好。'}, + {'id': '127268', 'name': '#740', + 'src': 'https://img3.xiji.com/images/19/01/12248f21bf2a429cce01b41fcfd79b232f391e70.jpg?1577773531#w', + 'rgb': (181, 46, 24), 'subtile': '脏橘色南瓜色百搭', 'desc': '网红人气爆款,实力显白,送人送礼佳品,这支口红真的是人见人爱,厚涂也可以hold住!'}, + {'id': '127269', 'name': '#888', + 'src': 'https://img1.xiji.com/images/19/01/003d9dd19d6612a42b107f427cea2a534c851c42.jpg?1577773531#w', + 'rgb': (220, 29, 36), 'subtile': '火焰开运色', 'desc': '888发发发,让人想到热烈的火焰,红红火火。如果觉得正红太艳丽大可选择这款,正红偏橘,非常显白有活力。'}, + {'id': '127270', 'name': '#999金属', + 'src': 'https://img4.xiji.com/images/19/01/fea134055f2050c0a3c9ad992529aa377b0a722f.jpg?1577773532#w', + 'rgb': (168, 15, 9), 'subtile': '人鱼姬正红', 'desc': '已经有999的小仙女一定不能错过这款金属光正红,偏光的微闪人鱼姬色在阳光下不灵不灵的,非常富有层次感。'}, + {'id': '127271', 'name': '#999滋润', + 'src': 'https://img0.xiji.com/images/19/01/6efcdf3646ab405e003d3feefa2d463ac20b9d3a.jpg?1577773534#w', + 'rgb': (201, 2, 5), 'subtile': '经典正红色', 'desc': '颜色最纯正的一款正红色,不挑肤色,喜庆特别显气质。嘴唇状态不好的小仙女一定要选这款,能让唇妆看起来更美腻~'}, + {'id': '127272', 'name': '#999哑光', + 'src': 'https://img1.xiji.com/images/19/01/ae4fa46845a4db70dc8fe02ed4faa214de12ad1c.jpg?1577773533#w', + 'rgb': (190, 18, 14), 'subtile': '经典正红色', 'desc': '李佳琦墙裂推荐的一个色号,每个女人都必须拥有,涂上气场两米八!哑光质地,不偏橘也不偏玫,厚涂薄涂都美到爆炸!'}] + :return:list, + """ + + lylj = xiji_spider.LYLJ() + ids, names, srcs = lylj.load_content() + print('Get data from network success!') + info = [] + for item_id, name, src in zip(ids, names, srcs): + info.append({'id': item_id, 'name': name, 'src': src}) + + item_detail = get_item_detail(settings.LYLJ_IMG_DIR) + + lylj_infos = utils.merge_iterables_of_dict('id', info, item_detail) + return lylj_infos + + +if __name__ == '__main__': + # ret = get_item_detail(settings.LYLJ_IMG_DIR) + ret = get_lylj_series() + print(ret) diff --git a/backend/core/xiji_spider.py b/backend/core/xiji_spider.py new file mode 100644 index 0000000..bb2e397 --- /dev/null +++ b/backend/core/xiji_spider.py @@ -0,0 +1,126 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# Created by Administrator at 2020/5/7 21:54 +import os +import re +import pathlib +from urllib.request import urlopen, Request + +from lxml import etree + +from backend import settings + + +class LYLJ: + """ + 烈艳蓝金系列唇膏爬虫 + """ + start_link = settings.DIOR_LYLJ_URL + + def __init__(self): + pass + + # @staticmethod + # def get_html(url): + # """ + # 爬取一次之后,保存网页到本地 + # """ + # content = Request(url, headers=settings.HEADERS) + # # content = urllib.request.urlopen(url) # 发出请求并且接收返回文本对象 + # response = urlopen(content, timeout=10) + # html = response.read() # 调用read()进行读取 + # with open('../../_data/xiji/products.html', 'wb+') as f: + # f.write(html) + # return html + + def current_item_id(self): + ret = re.split(r"[.-]", self.start_link) + return ret[-2] + + @staticmethod + def join_link(link): + """ + 组装成真实的link + :param link: + :return: + """ + return ''.join(['https:', link]) + + def mk_dir(self): + dir_name = '../../_data/lipsticks/Dior/{sn}'.format(sn=self.__class__.__name__.lower()) + if not os.path.exists(dir_name): + # [Python安全创建目录的方法](https://majing.io/posts/10000007281150) + pathlib.Path(dir_name).mkdir(parents=True, exist_ok=True) + return dir_name + + def load_content(self): + result = self.get_item_element_tree(self.start_link) + links = result.xpath('//*[@id="product_spec"]/ul/li/span[2]/ul/li/a/@href') + item_ids = result.xpath('//*[@id="product_spec"]/ul/li/span[2]/ul/li/a/@rel') + names = result.xpath('//*[@id="product_spec"]/ul/li/span[2]/ul/li/a/span') + + dir_name = self.mk_dir() + + img_src_list = [] # 对当前页单独处理 + img_src = self.get_image_src(result) + current_id = self.current_item_id() + self.save_image(img_src, dir_name, current_id) + img_src_list.append(img_src) + + current_names = [name.text for name in names] + links.pop(0) # 移除第一个script + print('Start get details……') + for img_id, link in zip(item_ids, links): + real_link = self.join_link(link) + result = self.get_item_element_tree(real_link) + img_src = self.get_image_src(result) + self.save_image(img_src, dir_name, img_id) + img_src_list.append(img_src) + + item_ids.insert(0, current_id) + return item_ids, current_names, img_src_list + + @staticmethod + def save_image(img_src, dir_name, image_id): + fp = '{dirn}/{fn}.{ext}'.format(dirn=dir_name, fn=str(image_id), ext='jpg') + with open(fp, 'wb') as f: + img = urlopen(img_src).read() + f.write(img) + return 0 + + def get_image_src(self, result_obj): + img_link = result_obj.xpath('//*[@id="op_product_zoom"]/img/@src') + real_img_src = self.join_link(img_link[0]) + return real_img_src + + @staticmethod + def get_item_element_tree(item_url): + """ + 给出网页地址,获得网页信息 + :param item_url: + :return: + """ + content = Request(item_url, headers=settings.HEADERS) + response = urlopen(content, timeout=10) + html = response.read() # 调用read()进行读取 + result = etree.HTML(html) + return result + + +if __name__ == '__main__': + lylj = LYLJ() + ret = lylj.load_content() + print(ret) + + ''' + (['127266', '127267', '127268', '127269', '127270', '127271', '127272'], + ['#520', '#080', '#740', '#888', '#999金属', '#999滋润', '#999哑光'], + ['https://img3.xiji.com/images/19/01/5b96671ae769403827ec84b8200805abb36a40a0.jpg?1577773530#w', + 'https://img0.xiji.com/images/19/01/cf3cc958da0bc2b3715b69038ad417ef29523110.jpg?1577773529#w', + 'https://img0.xiji.com/images/19/01/12248f21bf2a429cce01b41fcfd79b232f391e70.jpg?1577773531#w', + 'https://img2.xiji.com/images/19/01/003d9dd19d6612a42b107f427cea2a534c851c42.jpg?1577773531#w', + 'https://img2.xiji.com/images/19/01/fea134055f2050c0a3c9ad992529aa377b0a722f.jpg?1577773532#w', + 'https://img4.xiji.com/images/19/01/6efcdf3646ab405e003d3feefa2d463ac20b9d3a.jpg?1577773534#w', + 'https://img0.xiji.com/images/19/01/ae4fa46845a4db70dc8fe02ed4faa214de12ad1c.jpg?1577773533#w']) + + ''' diff --git a/backend/libs/baidu_api/LICENSE b/backend/libs/baidu_api/LICENSE new file mode 100644 index 0000000..8dada3e --- /dev/null +++ b/backend/libs/baidu_api/LICENSE @@ -0,0 +1,201 @@ + Apache License + Version 2.0, January 2004 + http://www.apache.org/licenses/ + + TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION + + 1. Definitions. + + "License" shall mean the terms and conditions for use, reproduction, + and distribution as defined by Sections 1 through 9 of this document. + + "Licensor" shall mean the copyright owner or entity authorized by + the copyright owner that is granting the License. + + "Legal Entity" shall mean the union of the acting entity and all + other entities that control, are controlled by, or are under common + control with that entity. For the purposes of this definition, + "control" means (i) the power, direct or indirect, to cause the + direction or management of such entity, whether by contract or + otherwise, or (ii) ownership of fifty percent (50%) or more of the + outstanding shares, or (iii) beneficial ownership of such entity. + + "You" (or "Your") shall mean an individual or Legal Entity + exercising permissions granted by this License. + + "Source" form shall mean the preferred form for making modifications, + including but not limited to software source code, documentation + source, and configuration files. + + "Object" form shall mean any form resulting from mechanical + transformation or translation of a Source form, including but + not limited to compiled object code, generated documentation, + and conversions to other media types. + + "Work" shall mean the work of authorship, whether in Source or + Object form, made available under the License, as indicated by a + copyright notice that is included in or attached to the work + (an example is provided in the Appendix below). + + "Derivative Works" shall mean any work, whether in Source or Object + form, that is based on (or derived from) the Work and for which the + editorial revisions, annotations, elaborations, or other modifications + represent, as a whole, an original work of authorship. For the purposes + of this License, Derivative Works shall not include works that remain + separable from, or merely link (or bind by name) to the interfaces of, + the Work and Derivative Works thereof. + + "Contribution" shall mean any work of authorship, including + the original version of the Work and any modifications or additions + to that Work or Derivative Works thereof, that is intentionally + submitted to Licensor for inclusion in the Work by the copyright owner + or by an individual or Legal Entity authorized to submit on behalf of + the copyright owner. For the purposes of this definition, "submitted" + means any form of electronic, verbal, or written communication sent + to the Licensor or its representatives, including but not limited to + communication on electronic mailing lists, source code control systems, + and issue tracking systems that are managed by, or on behalf of, the + Licensor for the purpose of discussing and improving the Work, but + excluding communication that is conspicuously marked or otherwise + designated in writing by the copyright owner as "Not a Contribution." + + "Contributor" shall mean Licensor and any individual or Legal Entity + on behalf of whom a Contribution has been received by Licensor and + subsequently incorporated within the Work. + + 2. Grant of Copyright License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + copyright license to reproduce, prepare Derivative Works of, + publicly display, publicly perform, sublicense, and distribute the + Work and such Derivative Works in Source or Object form. + + 3. Grant of Patent License. Subject to the terms and conditions of + this License, each Contributor hereby grants to You a perpetual, + worldwide, non-exclusive, no-charge, royalty-free, irrevocable + (except as stated in this section) patent license to make, have made, + use, offer to sell, sell, import, and otherwise transfer the Work, + where such license applies only to those patent claims licensable + by such Contributor that are necessarily infringed by their + Contribution(s) alone or by combination of their Contribution(s) + with the Work to which such Contribution(s) was submitted. If You + institute patent litigation against any entity (including a + cross-claim or counterclaim in a lawsuit) alleging that the Work + or a Contribution incorporated within the Work constitutes direct + or contributory patent infringement, then any patent licenses + granted to You under this License for that Work shall terminate + as of the date such litigation is filed. + + 4. Redistribution. You may reproduce and distribute copies of the + Work or Derivative Works thereof in any medium, with or without + modifications, and in Source or Object form, provided that You + meet the following conditions: + + (a) You must give any other recipients of the Work or + Derivative Works a copy of this License; and + + (b) You must cause any modified files to carry prominent notices + stating that You changed the files; and + + (c) You must retain, in the Source form of any Derivative Works + that You distribute, all copyright, patent, trademark, and + attribution notices from the Source form of the Work, + excluding those notices that do not pertain to any part of + the Derivative Works; and + + (d) If the Work includes a "NOTICE" text file as part of its + distribution, then any Derivative Works that You distribute must + include a readable copy of the attribution notices contained + within such NOTICE file, excluding those notices that do not + pertain to any part of the Derivative Works, in at least one + of the following places: within a NOTICE text file distributed + as part of the Derivative Works; within the Source form or + documentation, if provided along with the Derivative Works; or, + within a display generated by the Derivative Works, if and + wherever such third-party notices normally appear. The contents + of the NOTICE file are for informational purposes only and + do not modify the License. You may add Your own attribution + notices within Derivative Works that You distribute, alongside + or as an addendum to the NOTICE text from the Work, provided + that such additional attribution notices cannot be construed + as modifying the License. + + You may add Your own copyright statement to Your modifications and + may provide additional or different license terms and conditions + for use, reproduction, or distribution of Your modifications, or + for any such Derivative Works as a whole, provided Your use, + reproduction, and distribution of the Work otherwise complies with + the conditions stated in this License. + + 5. Submission of Contributions. Unless You explicitly state otherwise, + any Contribution intentionally submitted for inclusion in the Work + by You to the Licensor shall be under the terms and conditions of + this License, without any additional terms or conditions. + Notwithstanding the above, nothing herein shall supersede or modify + the terms of any separate license agreement you may have executed + with Licensor regarding such Contributions. + + 6. Trademarks. This License does not grant permission to use the trade + names, trademarks, service marks, or product names of the Licensor, + except as required for reasonable and customary use in describing the + origin of the Work and reproducing the content of the NOTICE file. + + 7. Disclaimer of Warranty. Unless required by applicable law or + agreed to in writing, Licensor provides the Work (and each + Contributor provides its Contributions) on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or + implied, including, without limitation, any warranties or conditions + of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A + PARTICULAR PURPOSE. You are solely responsible for determining the + appropriateness of using or redistributing the Work and assume any + risks associated with Your exercise of permissions under this License. + + 8. Limitation of Liability. In no event and under no legal theory, + whether in tort (including negligence), contract, or otherwise, + unless required by applicable law (such as deliberate and grossly + negligent acts) or agreed to in writing, shall any Contributor be + liable to You for damages, including any direct, indirect, special, + incidental, or consequential damages of any character arising as a + result of this License or out of the use or inability to use the + Work (including but not limited to damages for loss of goodwill, + work stoppage, computer failure or malfunction, or any and all + other commercial damages or losses), even if such Contributor + has been advised of the possibility of such damages. + + 9. Accepting Warranty or Additional Liability. While redistributing + the Work or Derivative Works thereof, You may choose to offer, + and charge a fee for, acceptance of support, warranty, indemnity, + or other liability obligations and/or rights consistent with this + License. However, in accepting such obligations, You may act only + on Your own behalf and on Your sole responsibility, not on behalf + of any other Contributor, and only if You agree to indemnify, + defend, and hold each Contributor harmless for any liability + incurred by, or claims asserted against, such Contributor by reason + of your accepting any such warranty or additional liability. + + END OF TERMS AND CONDITIONS + + APPENDIX: How to apply the Apache License to your work. + + To apply the Apache License to your work, attach the following + boilerplate notice, with the fields enclosed by brackets "{}" + replaced with your own identifying information. (Don't include + the brackets!) The text should be enclosed in the appropriate + comment syntax for the file format. We also recommend that a + file or class name and description of purpose be included on the + same "printed page" as the copyright notice for easier + identification within third-party archives. + + Copyright {yyyy} {name of copyright owner} + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. diff --git a/backend/libs/baidu_api/__init__.py b/backend/libs/baidu_api/__init__.py new file mode 100644 index 0000000..3af36ba --- /dev/null +++ b/backend/libs/baidu_api/__init__.py @@ -0,0 +1,5 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# Created by imoyao at 2020/5/2 9:31 + +# see also: https://ai.baidu.com/ai-doc/OCR/Ek3h7yeiq diff --git a/backend/libs/baidu_api/aip/__init__.py b/backend/libs/baidu_api/aip/__init__.py new file mode 100644 index 0000000..a2a3c0f --- /dev/null +++ b/backend/libs/baidu_api/aip/__init__.py @@ -0,0 +1,6 @@ +# -*- coding: utf-8 -*- +""" + aip public +""" + +from .ocr import AipOcr diff --git a/backend/libs/baidu_api/aip/base.py b/backend/libs/baidu_api/aip/base.py new file mode 100644 index 0000000..12b9a65 --- /dev/null +++ b/backend/libs/baidu_api/aip/base.py @@ -0,0 +1,278 @@ +# -*- coding: utf-8 -*- + +""" + AipBase +""" +import hmac +import json +import hashlib +import datetime +import base64 +import time +import sys +import requests +requests.packages.urllib3.disable_warnings() + + +if sys.version_info.major == 2: + from urllib import urlencode + from urllib import quote + from urlparse import urlparse +else: + from urllib.parse import urlencode + from urllib.parse import quote + from urllib.parse import urlparse + +class AipBase(object): + """ + AipBase + """ + + __accessTokenUrl = 'https://aip.baidubce.com/oauth/2.0/token' + + __reportUrl = 'https://aip.baidubce.com/rpc/2.0/feedback/v1/report' + + __scope = 'brain_all_scope' + + def __init__(self, appId, apiKey, secretKey): + """ + AipBase(appId, apiKey, secretKey) + """ + + self._appId = appId.strip() + self._apiKey = apiKey.strip() + self._secretKey = secretKey.strip() + self._authObj = {} + self._isCloudUser = None + self.__client = requests + self.__connectTimeout = 60.0 + self.__socketTimeout = 60.0 + self._proxies = {} + self.__version = '2_2_15' + + def getVersion(self): + """ + version + """ + return self.__version + + def setConnectionTimeoutInMillis(self, ms): + """ + setConnectionTimeoutInMillis + """ + + self.__connectTimeout = ms / 1000.0 + + def setSocketTimeoutInMillis(self, ms): + """ + setSocketTimeoutInMillis + """ + + self.__socketTimeout = ms / 1000.0 + + def setProxies(self, proxies): + """ + proxies + """ + + self._proxies = proxies + + def _request(self, url, data, headers=None): + """ + self._request('', {}) + """ + try: + result = self._validate(url, data) + if result != True: + return result + + authObj = self._auth() + params = self._getParams(authObj) + + data = self._proccessRequest(url, params, data, headers) + headers = self._getAuthHeaders('POST', url, params, headers) + response = self.__client.post(url, data=data, params=params, + headers=headers, verify=False, timeout=( + self.__connectTimeout, + self.__socketTimeout, + ), proxies=self._proxies + ) + obj = self._proccessResult(response.content) + + if not self._isCloudUser and obj.get('error_code', '') == 110: + authObj = self._auth(True) + params = self._getParams(authObj) + response = self.__client.post(url, data=data, params=params, + headers=headers, verify=False, timeout=( + self.__connectTimeout, + self.__socketTimeout, + ), proxies=self._proxies + ) + obj = self._proccessResult(response.content) + except (requests.exceptions.ReadTimeout, requests.exceptions.ConnectTimeout) as e: + return { + 'error_code': 'SDK108', + 'error_msg': 'connection or read data timeout', + } + + return obj + + def _validate(self, url, data): + """ + validate + """ + + return True + + def _proccessRequest(self, url, params, data, headers): + """ + 参数处理 + """ + + params['aipSdk'] = 'python' + params['aipVersion'] = self.__version + + return data + + def _proccessResult(self, content): + """ + formate result + """ + + if sys.version_info.major == 2: + return json.loads(content) or {} + else: + return json.loads(content.decode()) or {} + + def _auth(self, refresh=False): + """ + api access auth + """ + + #未过期 + if not refresh: + tm = self._authObj.get('time', 0) + int(self._authObj.get('expires_in', 0)) - 30 + if tm > int(time.time()): + return self._authObj + + obj = self.__client.get(self.__accessTokenUrl, verify=False, params={ + 'grant_type': 'client_credentials', + 'client_id': self._apiKey, + 'client_secret': self._secretKey, + }, timeout=( + self.__connectTimeout, + self.__socketTimeout, + ), proxies=self._proxies).json() + + self._isCloudUser = not self._isPermission(obj) + obj['time'] = int(time.time()) + self._authObj = obj + + return obj + + def _isPermission(self, authObj): + """ + check whether permission + """ + + scopes = authObj.get('scope', '') + + return self.__scope in scopes.split(' ') + + def _getParams(self, authObj): + """ + api request http url params + """ + + params = {} + + if self._isCloudUser == False: + params['access_token'] = authObj['access_token'] + + return params + + def _getAuthHeaders(self, method, url, params=None, headers=None): + """ + api request http headers + """ + + headers = headers or {} + params = params or {} + + if self._isCloudUser == False: + return headers + + urlResult = urlparse(url) + for kv in urlResult.query.strip().split('&'): + if kv: + k, v = kv.split('=') + params[k] = v + + # UTC timestamp + timestamp = datetime.datetime.utcnow().strftime('%Y-%m-%dT%H:%M:%SZ') + headers['Host'] = urlResult.hostname + headers['x-bce-date'] = timestamp + version, expire = '1', '1800' + + # 1 Generate SigningKey + val = "bce-auth-v%s/%s/%s/%s" % (version, self._apiKey, timestamp, expire) + signingKey = hmac.new(self._secretKey.encode('utf-8'), val.encode('utf-8'), + hashlib.sha256 + ).hexdigest() + + # 2 Generate CanonicalRequest + # 2.1 Genrate CanonicalURI + canonicalUri = quote(urlResult.path) + # 2.2 Generate CanonicalURI: not used here + # 2.3 Generate CanonicalHeaders: only include host here + + canonicalHeaders = [] + for header, val in headers.items(): + canonicalHeaders.append( + '%s:%s' % ( + quote(header.strip(), '').lower(), + quote(val.strip(), '') + ) + ) + canonicalHeaders = '\n'.join(sorted(canonicalHeaders)) + + # 2.4 Generate CanonicalRequest + canonicalRequest = '%s\n%s\n%s\n%s' % ( + method.upper(), + canonicalUri, + '&'.join(sorted(urlencode(params).split('&'))), + canonicalHeaders + ) + + # 3 Generate Final Signature + signature = hmac.new(signingKey.encode('utf-8'), canonicalRequest.encode('utf-8'), + hashlib.sha256 + ).hexdigest() + + headers['authorization'] = 'bce-auth-v%s/%s/%s/%s/%s/%s' % ( + version, + self._apiKey, + timestamp, + expire, + ';'.join(headers.keys()).lower(), + signature + ) + + return headers + + def report(self, feedback): + """ + 数据反馈 + """ + + data = {} + data['feedback'] = feedback + + return self._request(self.__reportUrl, data) + + def post(self, url, data, headers=None): + """ + self.post('', {}) + """ + + return self._request(url, data, headers) diff --git a/backend/libs/baidu_api/aip/ocr.py b/backend/libs/baidu_api/aip/ocr.py new file mode 100644 index 0000000..9816cfd --- /dev/null +++ b/backend/libs/baidu_api/aip/ocr.py @@ -0,0 +1,645 @@ +# -*- coding: utf-8 -*- + +""" +图像识别 +""" + +import math +import time +from .base import AipBase +from .base import base64 + + +class AipOcr(AipBase): + """ + 图像识别 + """ + + __generalBasicUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/general_basic' + + __accurateBasicUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/accurate_basic' + + __generalUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/general' + + __accurateUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/accurate' + + __generalEnhancedUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/general_enhanced' + + __webImageUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/webimage' + + __idcardUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/idcard' + + __bankcardUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/bankcard' + + __drivingLicenseUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/driving_license' + + __vehicleLicenseUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/vehicle_license' + + __licensePlateUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/license_plate' + + __businessLicenseUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/business_license' + + __receiptUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/receipt' + + __trainTicketUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/train_ticket' + + __taxiReceiptUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/taxi_receipt' + + __formUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/form' + + __tableRecognizeUrl = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/request' + + __tableResultGetUrl = 'https://aip.baidubce.com/rest/2.0/solution/v1/form_ocr/get_request_result' + + __vinCodeUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/vin_code' + + __quotaInvoiceUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/quota_invoice' + + __householdRegisterUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/household_register' + + __HKMacauExitentrypermitUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/HK_Macau_exitentrypermit' + + __taiwanExitentrypermitUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/taiwan_exitentrypermit' + + __birthCertificateUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/birth_certificate' + + __vehicleInvoiceUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/vehicle_invoice' + + __vehicleCertificateUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/vehicle_certificate' + + __invoiceUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/invoice' + + __airTicketUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/air_ticket' + + __insuranceDocumentsUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/insurance_documents' + + __vatInvoiceUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/vat_invoice' + + __qrcodeUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/qrcode' + + __numbersUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/numbers' + + __lotteryUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/lottery' + + __passportUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/passport' + + __businessCardUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/business_card' + + __handwritingUrl = 'https://aip.baidubce.com/rest/2.0/ocr/v1/handwriting' + + __customUrl = 'https://aip.baidubce.com/rest/2.0/solution/v1/iocr/recognise' + + def basicGeneral(self, image, options=None): + """ + 通用文字识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__generalBasicUrl, data) + + def basicGeneralUrl(self, url, options=None): + """ + 通用文字识别 + """ + options = options or {} + + data = {} + data['url'] = url + + data.update(options) + + return self._request(self.__generalBasicUrl, data) + + def basicAccurate(self, image, options=None): + """ + 通用文字识别(高精度版) + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__accurateBasicUrl, data) + + def general(self, image, options=None): + """ + 通用文字识别(含位置信息版) + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__generalUrl, data) + + def generalUrl(self, url, options=None): + """ + 通用文字识别(含位置信息版) + """ + options = options or {} + + data = {} + data['url'] = url + + data.update(options) + + return self._request(self.__generalUrl, data) + + def accurate(self, image, options=None): + """ + 通用文字识别(含位置高精度版) + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__accurateUrl, data) + + def enhancedGeneral(self, image, options=None): + """ + 通用文字识别(含生僻字版) + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__generalEnhancedUrl, data) + + def enhancedGeneralUrl(self, url, options=None): + """ + 通用文字识别(含生僻字版) + """ + options = options or {} + + data = {} + data['url'] = url + + data.update(options) + + return self._request(self.__generalEnhancedUrl, data) + + def webImage(self, image, options=None): + """ + 网络图片文字识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__webImageUrl, data) + + def webImageUrl(self, url, options=None): + """ + 网络图片文字识别 + """ + options = options or {} + + data = {} + data['url'] = url + + data.update(options) + + return self._request(self.__webImageUrl, data) + + def idcard(self, image, id_card_side, options=None): + """ + 身份证识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + data['id_card_side'] = id_card_side + + data.update(options) + + return self._request(self.__idcardUrl, data) + + def bankcard(self, image, options=None): + """ + 银行卡识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__bankcardUrl, data) + + def drivingLicense(self, image, options=None): + """ + 驾驶证识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__drivingLicenseUrl, data) + + def vehicleLicense(self, image, options=None): + """ + 行驶证识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__vehicleLicenseUrl, data) + + def licensePlate(self, image, options=None): + """ + 车牌识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__licensePlateUrl, data) + + def businessLicense(self, image, options=None): + """ + 营业执照识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__businessLicenseUrl, data) + + def receipt(self, image, options=None): + """ + 通用票据识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__receiptUrl, data) + + def trainTicket(self, image, options=None): + """ + 火车票识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__trainTicketUrl, data) + + def taxiReceipt(self, image, options=None): + """ + 出租车票识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__taxiReceiptUrl, data) + + def form(self, image, options=None): + """ + 表格文字识别同步接口 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__formUrl, data) + + def tableRecognitionAsync(self, image, options=None): + """ + 表格文字识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__tableRecognizeUrl, data) + + def getTableRecognitionResult(self, request_id, options=None): + """ + 表格识别结果 + """ + options = options or {} + + data = {} + data['request_id'] = request_id + + data.update(options) + + return self._request(self.__tableResultGetUrl, data) + + def vinCode(self, image, options=None): + """ + VIN码识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__vinCodeUrl, data) + + def quotaInvoice(self, image, options=None): + """ + 定额发票识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__quotaInvoiceUrl, data) + + def householdRegister(self, image, options=None): + """ + 户口本识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__householdRegisterUrl, data) + + def HKMacauExitentrypermit(self, image, options=None): + """ + 港澳通行证识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__HKMacauExitentrypermitUrl, data) + + def taiwanExitentrypermit(self, image, options=None): + """ + 台湾通行证识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__taiwanExitentrypermitUrl, data) + + def birthCertificate(self, image, options=None): + """ + 出生医学证明识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__birthCertificateUrl, data) + + def vehicleInvoice(self, image, options=None): + """ + 机动车销售发票识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__vehicleInvoiceUrl, data) + + def vehicleCertificate(self, image, options=None): + """ + 车辆合格证识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__vehicleCertificateUrl, data) + + def invoice(self, image, options=None): + """ + 税务局通用机打发票识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__invoiceUrl, data) + + def airTicket(self, image, options=None): + """ + 行程单识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__airTicketUrl, data) + + def insuranceDocuments(self, image, options=None): + """ + 保单识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__insuranceDocumentsUrl, data) + + def vatInvoice(self, image, options=None): + """ + 增值税发票识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__vatInvoiceUrl, data) + + def qrcode(self, image, options=None): + """ + 二维码识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__qrcodeUrl, data) + + def numbers(self, image, options=None): + """ + 数字识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__numbersUrl, data) + + def lottery(self, image, options=None): + """ + 彩票识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__lotteryUrl, data) + + def passport(self, image, options=None): + """ + 护照识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__passportUrl, data) + + def businessCard(self, image, options=None): + """ + 名片识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__businessCardUrl, data) + + def handwriting(self, image, options=None): + """ + 手写文字识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__handwritingUrl, data) + + def custom(self, image, options=None): + """ + 自定义模板文字识别 + """ + options = options or {} + + data = {} + data['image'] = base64.b64encode(image).decode() + + data.update(options) + + return self._request(self.__customUrl, data) + + def tableRecognition(self, image, options=None, timeout=10000): + """ + tableRecognition + """ + + result = self.tableRecognitionAsync(image) + + if 'error_code' in result: + return result + + requestId = result['result'][0]['request_id'] + for i in range(int(math.ceil(timeout / 1000.0))): + result = self.getTableRecognitionResult(requestId, options) + + # 完成 + if int(result['result']['ret_code']) == 3: + break + time.sleep(1) + + return result diff --git a/backend/libs/baidu_api/bin/aip_client b/backend/libs/baidu_api/bin/aip_client new file mode 100644 index 0000000..befec88 --- /dev/null +++ b/backend/libs/baidu_api/bin/aip_client @@ -0,0 +1,12 @@ +#/bin/bash + +FWDIR="$(dirname "$0")" +export PYTHONSTARTUP="$FWDIR/.baidu-aip-boostrap.py" +echo "# -*- coding: utf-8 -*-" > $PYTHONSTARTUP +echo "import sys" >> $PYTHONSTARTUP +echo "from aip import *" >> $PYTHONSTARTUP +echo "print '''" >> $PYTHONSTARTUP +echo " ------ welcome to use baidu aip, the best ai sdk! ------" >> $PYTHONSTARTUP +echo "'''" >> $PYTHONSTARTUP +echo "sys.ps1 = 'baidu-aip >>>'" >> $PYTHONSTARTUP +python \ No newline at end of file diff --git a/backend/settings.py b/backend/settings.py index b1163a5..766abfe 100644 --- a/backend/settings.py +++ b/backend/settings.py @@ -1,6 +1,8 @@ #!/usr/bin/env python # -*- coding: utf-8 -*- # Created by imoyao at 2020/4/12 21:52 +import os + JSON_LOAD_FP = '_data/Traditional-Chinese-Colors.json' YAML_LOAD_FP = '_data/colors-source.yml' NIPPON_COLOR_LOAD_FP = '_data/nippon-color.json' @@ -83,6 +85,7 @@ } # 因为黑白我们的算法基本可以识别所以此处不列出 REG_COLOR_SERES = r'\w*([灰|红|黄|绿|青|蓝|紫])\w*' +REG_LYLJ_SUBTITLE_EXP = r'.+\((.+)\)' # 'black', 'gray', 'white', 'red', 'yellow', 'green', 'cyan', 'blue', 'purple' COLOR_SERIES_MAP = { 'black': '黑', @@ -111,3 +114,15 @@ 'cmyk': [0, 59, 61, 28], 'desc': '朱砂的颜色,比大红活泼,也称铅朱朱色丹色(在YM对等的情况下,适量减少红色的成分就是该色的色彩系列感觉)' } + +TEST_IMAGE_FP = ['../../_data/lipsticks/999_zirun.jpg', '../../_data/lipsticks/999_metal.jpg'] +current_dir = os.path.dirname(os.path.abspath(__file__)) + +DIOR_LYLJ_URL = 'https://www.xiji.com/product-127266.html' +LYLJ_IMG_DIR = '../../_data/lipsticks/Dior/lylj' +HEADERS = { + 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/84.0.4121.0 Safari/537.36 Edg/84.0.495.2'} + + +def full_path(): + return os.path.join(current_dir, TEST_IMAGE_FP) diff --git a/backend/utils.py b/backend/utils.py new file mode 100644 index 0000000..507575a --- /dev/null +++ b/backend/utils.py @@ -0,0 +1,23 @@ +#!/usr/bin/env python +# -*- coding: utf-8 -*- +# Created by Administrator at 2020/5/10 9:51 +from collections import defaultdict +import itertools + + +def merge_iterables_of_dict(shared_key, *iterables): + """ + see also:[🐍PyTricks | Python 中如何合并一个内字典列表? | 别院牧志](https://imoyao.github.io/blog/2020-04-19/python-merge-two-list-of-dicts/) + chinese_colors_data 放前面,因为有描述和图片 + :param shared_key: + :param iterables: + :return: + """ + result = defaultdict(dict) + for dictionary in itertools.chain.from_iterable(iterables): + result[dictionary[shared_key]].update(dictionary) + # for dictionary in result.values(): + # dictionary.pop(shared_key) + # return result + result = list(result.values()) # 保证返回为list,否则:TypeError: Object of type dict_values is not JSON serializable + return result diff --git a/src/main.js b/src/main.js index 697d35d..ada1058 100644 --- a/src/main.js +++ b/src/main.js @@ -2,7 +2,7 @@ import Vue from 'vue' import App from './App.vue' import router from './router' import store from './store' -// import ElementUI from 'element-ui' +import ElementUI from 'element-ui' import 'normalize.css' import 'element-ui/lib/theme-chalk/index.css' @@ -15,7 +15,7 @@ import commons from './commons.js' Vue.prototype.common = commons Vue.config.productionTip = false -// Vue.use(ElementUI) +Vue.use(ElementUI) new Vue({ router, diff --git a/src/views/BasePage.vue b/src/views/BasePage.vue index ee803eb..0f860af 100644 --- a/src/views/BasePage.vue +++ b/src/views/BasePage.vue @@ -53,6 +53,7 @@ class="n" type="circle"/>
+ @@ -346,6 +347,19 @@ export default { letter-spacing: 0.5rem; font-size: 3rem; } + .lip-img{ + position: absolute; + bottom: 15rem; + left: 10rem; + right: 8rem; + @include for-phone { + display:none; + } + @include for-tablet { + bottom: 25rem; + right: 5rem; + } + } .color-decs { font-family: 'FZQKBYSJT',serif; position: absolute; diff --git a/src/views/Lipsticks.vue b/src/views/Lipsticks.vue index aa21da0..42ae932 100644 --- a/src/views/Lipsticks.vue +++ b/src/views/Lipsticks.vue @@ -12,6 +12,18 @@ class="select-brand" @change="handleChange"/> +
+
+ +
+ +
@@ -45,6 +57,12 @@ export default { lipData: [], selectedLipstrik: [], colorSet: new Set(), + fits: ['cover'], + url: '//img4.xiji.com/images/19/01/6efcdf3646ab405e003d3feefa2d463ac20b9d3a.jpg', + srcList: [ + '//ci.xiaohongshu.com/26cd24f8-1440-5c74-a82a-fcbee522bade?imageView2/2/w/1080/format/jpg', + '//ci.xiaohongshu.com/7d62a4a7-2006-4aeb-a3f7-050f6152e1f5@r_1280w_1280h.jpg', + ], } }, created () { @@ -108,6 +126,7 @@ export default {