EADST

Python: Obtain Baidu Images Using Web Crawler

Python: Obtain Baidu Images Using Web Crawler.

Here is the main code.

# -- coding: utf-8 --
import os
import re
import time
import requests

class CarCollect():

def __init__(self, path='./name.txt'):
    self.num = 1
    self.class_number = 0
    self.line_list = []
    with open(path, encoding='utf-8') as file:
        self.line_list = [k.strip() for k in file.readlines()]
        self.class_number = int(self.line_list[0])
        self.line_list = self.line_list[1:]

def dowmload_picture(self, html, keyword, save_path):
    pic_url = re.findall('"objURL":"(.*?)",', html, re.S)  # get image url
    print('Finding keyword: ' + keyword + ' images, start downloading...')
    for each in pic_url:
        print('='*60)
        print('Downloading ' + keyword + ' number ' + str(self.num) + ' image, url: ' + str(each))
        try:
            if each:
                pic = requests.get(each, timeout=7)
                string = save_path + r'\\' + keyword + '_' + str(self.num) + '.jpg'
                if len(pic.content) > 10000: # img size > 10k
                    with open(string, 'wb') as fp:
                        fp.write(pic.content)
                        self.num += 1
        except BaseException:
            print('error, cannot download')
        if self.num > self.class_number:
            break

def __call__(self):
    headers = {
        'Accept-Language': 'zh-CN,zh;q=0.8,zh-TW;q=0.7,zh-HK;q=0.5,en-US;q=0.3,en;q=0.2',
        'Connection': 'keep-alive',
        'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:60.0) Gecko/20100101 Firefox/60.0',
        'Upgrade-Insecure-Requests': '1'
    }
    session = requests.Session()
    session.headers = headers

    for word in self.line_list:
        # create a folder
        save_path = word + '_file'
        time_now = time.strftime("%Y%m%d_%H%M%S", time.localtime())
        save_path += "_" + time_now
        os.mkdir(save_path)
        # get images
        url = 'https://image.baidu.com/search/flip?tn=baiduimage&ie=utf-8&word=' + word + '&pn='
        image_number = 0
        self.num = 1
        while image_number < self.class_number:
            try:
                result = session.get(url + str(image_number), timeout=10, allow_redirects=False)
                self.dowmload_picture(result.text, word, save_path)
            except:
                print('Internet error')
            image_number += 60

if name == 'main': path = './keywords.txt' car_collect = CarCollect(path) car_collect() print('Done.')

Here is the text file, keywords.txt. The first line is the number we want to obtain from each keyword. The following lines are the keywords.

20
Dog
Cat

相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
HuggingFace uWSGI Password Pandas 签证 Hungarian Safetensors Qwen2 Vim 图形思考法 torchinfo Baidu TensorFlow OCR git-lfs Bin Github tqdm CTC Base64 净利润 Card ONNX PDB RGB OpenCV GGML NLTK Jetson Bipartite GPTQ TSV Land DeepSeek Bitcoin XGBoost 强化学习 Agent Zip DeepStream CC 关于博主 NameSilo Google Attention SAM AI CV Nginx JSON Hilton FP64 Python LeetCode News Tensor Excel v0.dev Mixtral GIT 算法题 Permission Food Shortcut 阿里云 Rebuttal UNIX LoRA VGG-16 Cloudreve FP16 EXCEL Template Paper Datetime LLAMA MD5 SQL llama.cpp NLP Miniforge VPN 搞笑 腾讯云 CEIR Jupyter Image2Text PyTorch 继承 git ms-swift SPIE Augmentation PDF hf ChatGPT Docker Plate Clash QWEN 递归学习法 diffusers LaTeX RAR Windows Pillow Interview CUDA TensorRT Statistics 公式 Streamlit VSCode InvalidArgumentError Sklearn 飞书 音频 Qwen2.5 多线程 BF16 transformers Qwen 证件照 Data Bert Markdown Git 论文速读 Search Proxy Quantize Diagram Breakpoint Algorithm 论文 Firewall Ptyhon Crawler Linux Gemma Tiktoken WAN Video Disk Logo Hotel Llama logger Quantization SQLite CAM Vmess Magnet YOLO Translation Dataset GoogLeNet Paddle 财报 域名 Numpy WebCrawler FlashAttention UI Domain 报税 HaggingFace BTC RL API Web Heatmap 顶会 FP8 Knowledge Anaconda 图标 Michelin IndexTTS2 Freesound PyCharm Claude 云服务器 第一性原理 TTS SVR Animate ResNet-50 Math FP32 Django BeautifulSoup Color tar LLM CSV Pickle COCO Review Plotly CLAP GPT4 FastAPI printf Input Tracking C++ scipy Random uwsgi icon v2ray Website Ubuntu XML Distillation PIP Transformers Pytorch OpenAI 版权 mmap Conda ModelScope Use 多进程
站点统计

本站现有博文334篇,共被浏览933482

本站已经建立2643天!

热门文章
文章归档
回到顶部