python自動(dòng)從arxiv下載paper的示例代碼
#!/usr/bin/env python # -*- coding: utf-8 -*- # @Time : 2020/02/11 21:44 # @Author : dangxusheng # @Email : dangxusheng163@163.com # @File : download_by_href.py ''' 自動(dòng)從arxiv.org 下載文獻(xiàn) ''' import os import os.path as osp import requests from lxml import etree from pprint import pprint import re import time import glob headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/80.0.3987.87 Safari/537.36", "Host": 'arxiv.org' } HREF_CN = 'http://cn.arxiv.org/pdf/' HREF_SRC = 'http://cn.arxiv.org/pdf/' SAVE_PATH = '/media/dangxs/E/Paper/download_at_20200730' os.makedirs(SAVE_PATH, exist_ok=True) FAIL_URLS = [] FAIL_URLS_TXT = f'{SAVE_PATH}/fail_urls.txt' def download(url, title): pattern = r'[\\/:*?"\'<>|\r\n]+' new_title = re.sub(pattern, " ", title) print(f'new title: {new_title}') save_filepath = '%s/%s.pdf' % (SAVE_PATH, new_title) if osp.exists(save_filepath) and osp.getsize(save_filepath) > 50 * 1024: print(f'this pdf is be existed.') return True try: with open(save_filepath, 'wb') as file: # 分字節(jié)下載 r = requests.get(url, stream=True, timeout=None) for i in r.iter_content(2048): file.write(i) if osp.getsize(save_filepath) >= 10 * 1024: print('%s 下載成功.' % title) return True except Exception as e: print(e) return False # 從arxiv.org 去下載 def search(start_size=0, title_keywords='Facial Expression'): # 訪問(wèn)地址: https://arxiv.org/find/grp_eess,grp_stat,grp_cs,grp_econ,grp_math/1/ti:+Face/0/1/0/past,2018,2019/0/1?skip=200&query_id=1c582e6c8afc6146&client_host=cn.arxiv.org req_url = 'https://arxiv.org/search/advanced' req_data = { 'advanced': 1, 'terms-0-operator': 'AND', 'terms-0-term': title_keywords, 'terms-0-field': 'title', 'classification-computer_science': 'y', 'classification-physics_archives': 'all', 'classification-include_cross_list': 'include', 'date-filter_by': 'date_range', # date_range | specific_year # 'date-year': DOWN_YEAR, 'date-year': '', 'date-from_date': '2015', 'date-to_date': '2020', 'date-date_type': 'announced_date_first', # submitted_date | submitted_date_first | announced_date_first 'abstracts': 'show', 'size': 50, 'order': '-announced_date_first', 'start': start_size, } res = requests.get(req_url, params=req_data, headers=headers) html = res.content.decode() html = etree.HTML(html) total_text = html.xpath('//h1[@class="title is-clearfix"]/text()') total_text = ''.join(total_text).replace('\n', '').lstrip(' ').strip(' ') # i.e. : Showing 1–50 of 355 results num = re.findall('\d+', total_text) # Sorry, your query returned no results if len(num) == 0: return [], 0 total = int(num[-1]) # 查詢總條數(shù) paper_list = html.xpath('//ol[@class="breathe-horizontal"]/li') info_list = [] for p in paper_list: title = p.xpath('./p[@class="title is-5 mathjax"]//text()') title = ''.join(title).replace('\n', '').lstrip(' ').strip(' ') href = p.xpath('./div/p/a/@href')[0] info_list.append({'title': title, 'href': href}) return info_list, total # 去指定頁(yè)面下載 def search_special(): res = requests.get('https://gitee.com/weberyoung/the-gan-zoo?_from=gitee_search') html = res.content.decode() html = etree.HTML(html) paper_list = html.xpath('//div[@class="file_content markdown-body"]//li') info_list = [] for p in paper_list: title = p.xpath('.//text()') title = ''.join(title).replace('\n', '').lstrip(' ').strip(' ') href = p.xpath('./a/@href')[0] info_list.append({'title': title, 'href': href}) pprint(info_list) return info_list if __name__ == '__main__': page_idx = 0 total = 1000 keywords = 'Facial Action Unit' while page_idx <= total // 50: paper_list, total = search(page_idx * 50, keywords) print(f'total: {total}') if total == 0: print('no found .') exit(0) for p in paper_list: title = p['title'] href = HREF_CN + p['href'].split('/')[-1] + '.pdf' print(href) if not download(href, title): print('從國(guó)內(nèi)鏡像下載失敗,從源地址開(kāi)始下載 >>>>') # 使用國(guó)際URL再下載一次 href = HREF_SRC + p['href'].split('/')[-1] + '.pdf' if not download(href, title): FAIL_URLS.append(p) page_idx += 1 # 下載最后的部分 last_1 = total - page_idx * 50 paper_list, total = search(last_1, keywords) for p in paper_list: title = p['title'] href = HREF_CN + p['href'].split('/')[-1] + '.pdf' if not download(href, title): FAIL_URLS.append(p) time.sleep(1) pprint(FAIL_URLS) with open(FAIL_URLS_TXT, 'a+') as f: for item in FAIL_URLS: href = item['href'] title = item['title'] f.write(href + '\n') print('done.')
以上就是python自動(dòng)從arxiv下載paper的示例代碼的詳細(xì)內(nèi)容,更多關(guān)于python 從arxiv下載paper的資料請(qǐng)關(guān)注腳本之家其它相關(guān)文章!
相關(guān)文章
淺談Python 敏感詞過(guò)濾的實(shí)現(xiàn)
這篇文章主要介紹了淺談Python 敏感詞過(guò)濾的實(shí)現(xiàn),文中通過(guò)示例代碼介紹的非常詳細(xì),對(duì)大家的學(xué)習(xí)或者工作具有一定的參考學(xué)習(xí)價(jià)值,需要的朋友們下面隨著小編來(lái)一起學(xué)習(xí)學(xué)習(xí)吧2019-08-08python 子類調(diào)用父類的構(gòu)造函數(shù)實(shí)例
這篇文章主要介紹了python 子類調(diào)用父類的構(gòu)造函數(shù)實(shí)例,具有很好的參考價(jià)值,希望對(duì)大家有所幫助。一起跟隨小編過(guò)來(lái)看看吧2020-03-03Pandas計(jì)算元素的數(shù)量和頻率的方法(出現(xiàn)的次數(shù))
本文主要介紹了Pandas計(jì)算元素的數(shù)量和頻率的方法,文中通過(guò)示例代碼介紹的非常詳細(xì),對(duì)大家的學(xué)習(xí)或者工作具有一定的參考學(xué)習(xí)價(jià)值,需要的朋友們下面隨著小編來(lái)一起學(xué)習(xí)學(xué)習(xí)吧2023-02-02PyQt5實(shí)現(xiàn)QLineEdit添加clicked信號(hào)的方法
今天小編就為大家分享一篇PyQt5實(shí)現(xiàn)QLineEdit添加clicked信號(hào)的方法,具有很好的參考價(jià)值,希望對(duì)大家有所幫助。一起跟隨小編過(guò)來(lái)看看吧2019-06-06在python下使用tensorflow判斷是否存在文件夾的實(shí)例
今天小編就為大家分享一篇在python下使用tensorflow判斷是否存在文件夾的實(shí)例,具有很好的參考價(jià)值,希望對(duì)大家有所幫助。一起跟隨小編過(guò)來(lái)看看吧2019-06-06Win10搭建Pyspark2.4.4+Pycharm開(kāi)發(fā)環(huán)境的圖文教程(親測(cè))
本文主要介紹了Win10搭建Pyspark2.4.4+Pycharm開(kāi)發(fā)環(huán)境的圖文教程(親測(cè)),文中通過(guò)示例代碼介紹的非常詳細(xì),對(duì)大家的學(xué)習(xí)或者工作具有一定的參考學(xué)習(xí)價(jià)值,需要的朋友們下面隨著小編來(lái)一起學(xué)習(xí)學(xué)習(xí)吧2023-02-02如何通過(guò)pycharm實(shí)現(xiàn)對(duì)數(shù)據(jù)庫(kù)的查詢等操作(非多步操作)
這篇文章主要介紹了如何通過(guò)pycharm實(shí)現(xiàn)對(duì)數(shù)據(jù)庫(kù)的查詢等操作(非多步操作),具有很好的參考價(jià)值,希望對(duì)大家有所幫助。如有錯(cuò)誤或未考慮完全的地方,望不吝賜教2022-07-07