EADST

Python: Obtain Baidu Images Using Web Crawler

Python: Obtain Baidu Images Using Web Crawler.

Here is the main code.

# -- coding: utf-8 --
import os
import re
import time
import requests

class CarCollect():

def __init__(self, path='./name.txt'):
    self.num = 1
    self.class_number = 0
    self.line_list = []
    with open(path, encoding='utf-8') as file:
        self.line_list = [k.strip() for k in file.readlines()]
        self.class_number = int(self.line_list[0])
        self.line_list = self.line_list[1:]

def dowmload_picture(self, html, keyword, save_path):
    pic_url = re.findall('"objURL":"(.*?)",', html, re.S)  # get image url
    print('Finding keyword: ' + keyword + ' images, start downloading...')
    for each in pic_url:
        print('='*60)
        print('Downloading ' + keyword + ' number ' + str(self.num) + ' image, url: ' + str(each))
        try:
            if each:
                pic = requests.get(each, timeout=7)
                string = save_path + r'\\' + keyword + '_' + str(self.num) + '.jpg'
                if len(pic.content) > 10000: # img size > 10k
                    with open(string, 'wb') as fp:
                        fp.write(pic.content)
                        self.num += 1
        except BaseException:
            print('error, cannot download')
        if self.num > self.class_number:
            break

def __call__(self):
    headers = {
        'Accept-Language': 'zh-CN,zh;q=0.8,zh-TW;q=0.7,zh-HK;q=0.5,en-US;q=0.3,en;q=0.2',
        'Connection': 'keep-alive',
        'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64; rv:60.0) Gecko/20100101 Firefox/60.0',
        'Upgrade-Insecure-Requests': '1'
    }
    session = requests.Session()
    session.headers = headers

    for word in self.line_list:
        # create a folder
        save_path = word + '_file'
        time_now = time.strftime("%Y%m%d_%H%M%S", time.localtime())
        save_path += "_" + time_now
        os.mkdir(save_path)
        # get images
        url = 'https://image.baidu.com/search/flip?tn=baiduimage&ie=utf-8&word=' + word + '&pn='
        image_number = 0
        self.num = 1
        while image_number < self.class_number:
            try:
                result = session.get(url + str(image_number), timeout=10, allow_redirects=False)
                self.dowmload_picture(result.text, word, save_path)
            except:
                print('Internet error')
            image_number += 60

if name == 'main': path = './keywords.txt' car_collect = CarCollect(path) car_collect() print('Done.')

Here is the text file, keywords.txt. The first line is the number we want to obtain from each keyword. The following lines are the keywords.

20
Dog
Cat

相关标签
About Me
XD
Goals determine what you are going to be.
Category
标签云
Bitcoin RL PDF EXCEL GGML Dataset mmap Math Distillation Domain WAN BTC Pandas Disk Heatmap Translation GPTQ CV Safetensors Freesound Land WebCrawler Website AI BF16 NLTK VGG-16 RGB Docker GIT Algorithm COCO Python Knowledge Animate Jev 算法题 第一性原理 TensorRT Card CLAP VPN Breakpoint 图标 NameSilo v0.dev Tensor CC RAR API Qwen uwsgi Anaconda scipy FlashAttention git-lfs Quantization Augmentation OpenCV PDB FastAPI XML diffusers Statistics ms-swift TensorFlow Password DeepStream SQLite Pytorch Rebuttal Git icon SVR Qwen2.5 Video Michelin Interview LLM CTC DeepSeek Input Bin CSV Plotly 图形思考法 多进程 Excel Baidu TTS 递归学习法 Color Agent Jupyter ONNX Diagram InvalidArgumentError LLAMA hf SPIE Logo Image2Text Data Linux Vim 报税 域名 uWSGI JSON Numpy Tracking Miniforge printf CAM Search Hotel Random Template Zip Plate FP8 LoRA Datetime logger Harness Transformers llama.cpp GoogLeNet 公式 关于博主 PyCharm FP16 飞书 Github FP32 Markdown 版权 Bipartite ResNet-50 签证 强化学习 Qwen2 Proxy C++ BeautifulSoup VSCode 证件照 tqdm ChatGPT HuggingFace FP64 Ptyhon HaggingFace Firewall Magnet 净利润 Pillow Attention Conda SQL Crawler Permission IndexTTS2 音频 GPT4 Mixtral Llama UI News git 论文 OCR YOLO Bert Cloudreve Base64 QWEN Review API网关 Hilton LaTeX Tiktoken CEIR UNIX Clash 论文速读 PIP XGBoost Ubuntu 腾讯云 LeetCode Web 财报 transformers Nginx Sklearn Jetson 阿里云 Claude ModelScope Paper Streamlit tar 顶会 MD5 云服务器 SAM NLP Hungarian 多线程 torchinfo Pickle Use 继承 Food 搞笑 CUDA Windows Gemma Django Paddle PyTorch v2ray Quantize Shortcut TSV OpenAI Vmess Google
站点统计

本站现有博文337篇,共被浏览961716次

本站已经建立2673天!

热门文章
文章归档
回到顶部