java多线程的爬虫撸177图片~~

阅读: 评论:0

java多线程的爬虫撸177图片~~

java多线程的爬虫撸177图片~~

#来个多线程的

主程序
package cn.fu.threadimage;import org.jsoup.Jsoup;
import des.Document;
import des.Element;
import org.jsoup.select.Elements;import java.io.File;
import java.io.IOException;
import java.URL;
import urrent.ExecutorService;
import urrent.Executors;public class CrawImage {static String url = ".html"+"/";static String file;//下载存放路径static Integer num;//分页static Integer subcut;//图片命名截取imgurl,仅仅针对当前网站适用;static {try {Document document = Jsoup.parse(new URL(url), 5000);//获取标题Element element = ElementsByClass("entry-title").first();file = "F://paqu/" + ();//判断目标文件夹是否存在File files = new File(file);if (!ists()) {files.mkdirs();}Elements select = document.select(".page-links>a");//获取分页num = select.size();//177pic vpn访问网:www.177pic.pw 内网:www.177picaa.pwif (ains("aa")) {subcut = 40;} else {subcut = 38;}} catch (IOException e) {e.printStackTrace();}}public static void main(String[] args) throws Exception {try {//创建一个缓冲池ExecutorService pool = wCachedThreadPool();//设置其容量为9pool = wFixedThreadPool(9);for (int i = 1; i < num; i++) {//获取指定网页源码Document document = t(url + i).userAgent("Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.31 (KHTML, like Gecko) Chrome/26.0.1410.64 Safari/537.31").get();Elements pages = document.select(".page-links>a");getUrl(document, pool);}pool.shutdown();} catch (Exception e) {System.out.print(e);}}public static void getUrl(Document document, ExecutorService pool) {Elements elements = ElementsByClass("alignnone");for (Element el : elements) {String imageUrl = el.attr("data-lazy-src");if (imageUrl != "") {//下载图片ute(new DownloadImage(imageUrl, file, subcut));System.out.println(imageUrl);}}}
}
下载工具
package cn.fu.threadimage;import java.io.*;
import java.HttpURLConnection;
import java.MalformedURLException;
import java.URL;public class DownloadImage implements Runnable {String file;//下载的目标路径String downUrl;int subcut;public DownloadImage(String downUrl, String file,int subcut) {this.downUrl = downUrl;this.file = file;this.subcut=subcut;}public void run() {InputStream is;FileOutputStream out;try {URL url = new URL(downUrl);HttpURLConnection connection = (HttpURLConnection) url.openConnection();connection.setRequestProperty("User-Agent", "Mozilla/4.0 (compatible; MSIE 5.0; Windows NT; DigExt)");is = InputStream();// 创建文件File fileofImg = new File(file + "/" + downUrl.substring(subcut));out = new FileOutputStream(fileofImg);int i = 0;while ((i = is.read()) != -1) {out.write(i);}is.close();out.close();} catch (MalformedURLException e) {// TODO Auto-generated catch blocke.printStackTrace();} catch (FileNotFoundException e) {// TODO Auto-generated catch blocke.printStackTrace();} catch (IOException e) {// TODO Auto-generated catch blocke.printStackTrace();}}
}

本文发布于:2024-01-28 09:57:55,感谢您对本站的认可!

本文链接:https://www.4u4v.net/it/17064070816609.html

版权声明:本站内容均来自互联网,仅供演示用,请勿用于商业和其他非法用途。如果侵犯了您的权益请与我们联系,我们将在24小时内删除。

标签:爬虫   多线程   图片   java
留言与评论(共有 0 条评论)
   
验证码:

Copyright ©2019-2022 Comsenz Inc.Powered by ©

网站地图1 网站地图2 网站地图3 网站地图4 网站地图5 网站地图6 网站地图7 网站地图8 网站地图9 网站地图10 网站地图11 网站地图12 网站地图13 网站地图14 网站地图15 网站地图16 网站地图17 网站地图18 网站地图19 网站地图20 网站地图21 网站地图22/a> 网站地图23