EADST

Python: Obtain Baidu Images Using Web Crawler

Python: Obtain Baidu Images Using Web Crawler.

Here is the main code.

# -- coding: utf-8 --
import os
import re
import time
import requests

class CarCollect():

def __init__(self, path='./name.txt'):
    self.num = 1
    self.class_number = 0
    self.line_list = []
    with open(path, encoding='utf-8') as file:
        self.line_list = [k.strip() for k in file.readlines()]
        self.class_number = int(self.line_list[0])
        self.line_list = self.line_list[1:]

def dowmload_picture(self, html, keyword, save_path):
    pic_url = re.findall('"objURL":"(.*?)",', html, re.S)  # get image url
    print('Finding keyword: ' + keyword + ' images, start downloading...')
    for each in pic_url:
        print('='*60)
        print('Downloading ' + keyword + ' number ' + str(self.num) + ' image, url: ' + str(each))
        try:
            if each:
                pic = requests.get(each, timeout=7)
                string = save_path + r'\\' + keyword + '_' + str(self.num) + '.jpg'
                if len(pic.content) > 10000: # img size > 10k
                    with open(string, 'wb') as fp:
                        fp.write(pic.content)
                        self.num += 1
        except BaseException:
            print('error, cannot download')
        if self.num > self.class_number:
            break

def __call__(self):
    headers = {
        'Accept-Language': 'zh-CN,zh;q=0.8,zh-TW;q=0.7,zh-HK;q=0.5,en-US;q=0.3,en;q=0.2',
        'Connection': 'keep-alive',
        'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:60.0) Gecko/20100101 Firefox/60.0',
        'Upgrade-Insecure-Requests': '1'
    }
    session = requests.Session()
    session.headers = headers

    for word in self.line_list:
        # create a folder
        save_path = word + '_file'
        time_now = time.strftime("%Y%m%d_%H%M%S", time.localtime())
        save_path += "_" + time_now
        os.mkdir(save_path)
        # get images
        url = 'https://image.baidu.com/search/flip?tn=baiduimage&ie=utf-8&word=' + word + '&pn='
        image_number = 0
        self.num = 1
        while image_number < self.class_number:
            try:
                result = session.get(url + str(image_number), timeout=10, allow_redirects=False)
                self.dowmload_picture(result.text, word, save_path)
            except:
                print('Internet error')
            image_number += 60

if name == 'main': path = './keywords.txt' car_collect = CarCollect(path) car_collect() print('Done.')

Here is the text file, keywords.txt. The first line is the number we want to obtain from each keyword. The following lines are the keywords.

20
Dog
Cat

相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
腾讯云 SPIE TTS Jupyter YOLO Hungarian torchinfo Tensor ResNet-50 Permission Conda Qwen2 财报 FP32 Ubuntu Michelin MD5 Translation 域名 论文 Git News Statistics Breakpoint Interview Proxy git FlashAttention FastAPI Datetime GPTQ tqdm AI v0.dev BTC Vmess Attention Ptyhon RL Animate Food Linux COCO Gemma Bert 搞笑 多线程 PDF WebCrawler Hilton Excel Pillow Mixtral icon XGBoost Llama Vim 强化学习 净利润 版权 Tiktoken Dataset HaggingFace CEIR InvalidArgumentError Claude 云服务器 SQLite transformers 证件照 PyTorch uWSGI Freesound Google SVR Markdown 继承 C++ 音频 TSV Use Agent Bin mmap Augmentation DeepStream BeautifulSoup LLAMA LaTeX Plotly Quantization printf Cloudreve Image2Text Crawler 多进程 Anaconda SAM 签证 Website HuggingFace Hotel CUDA API Jetson PIP GoogLeNet WAN Shortcut TensorFlow Bipartite Nginx NameSilo Distillation Rebuttal 顶会 Miniforge llama.cpp 递归学习法 LLM NLTK git-lfs UNIX Template 图标 VPN Heatmap Baidu CSV 阿里云 QWEN Tracking Input Random Card Clash diffusers Windows Paper EXCEL Python Pytorch GPT4 DeepSeek Pickle 第一性原理 CC LeetCode IndexTTS2 v2ray Transformers GIT FP8 Domain Streamlit FP16 RGB 图形思考法 Password BF16 Logo VSCode Numpy logger FP64 ms-swift Magnet 算法题 Plate OpenAI CAM Github Disk Math Color Data ChatGPT 飞书 Django PDB Base64 ONNX UI Knowledge Docker 关于博主 Qwen NLP Diagram RAR JSON Sklearn Land hf Review Search Firewall uwsgi 报税 Web PyCharm Zip LoRA ModelScope CLAP Video OCR VGG-16 OpenCV CV SQL Safetensors scipy CTC GGML Pandas Bitcoin 论文速读 TensorRT tar Algorithm Qwen2.5 公式 Quantize Paddle XML
站点统计

本站现有博文333篇,共被浏览914721

本站已经建立2618天!

热门文章
文章归档
回到顶部