diff --git a/src/pa/dangdang/Book.java b/src/pa/dangdang/Book.java new file mode 100644 index 0000000..efc5072 --- /dev/null +++ b/src/pa/dangdang/Book.java @@ -0,0 +1,138 @@ +package pa.dangdang; + +public class Book { + + private String title; + private String author; // 作者 + private String publisher; // 出版社 + private double oldprice; // 出版时间 + private double newprice; // 价格 + private String href; // 图书详情url + + public Book() { + } + + public Book(String title, String author, String publisher, double oldprice, double newprice, String href) { + this.title = title; + this.author = author; + this.publisher = publisher; + this.oldprice = oldprice; + this.newprice = newprice; + this.href = href; + } + + /** + * 获取 + * + * @return title + */ + public String getTitle() { + return title; + } + + /** + * 设置 + * + * @param title + */ + public void setTitle(String title) { + this.title = title; + } + + /** + * 获取 + * + * @return author + */ + public String getAuthor() { + return author; + } + + /** + * 设置 + * + * @param author + */ + public void setAuthor(String author) { + this.author = author; + } + + /** + * 获取 + * + * @return publisher + */ + public String getPublisher() { + return publisher; + } + + /** + * 设置 + * + * @param publisher + */ + public void setPublisher(String publisher) { + this.publisher = publisher; + } + + /** + * 获取 + * + * @return oldprice + */ + public double getOldprice() { + return oldprice; + } + + /** + * 设置 + * + * @param oldprice + */ + public void setOldprice(double oldprice) { + this.oldprice = oldprice; + } + + /** + * 获取 + * + * @return newprice + */ + public double getNewprice() { + return newprice; + } + + /** + * 设置 + * + * @param newprice + */ + public void setNewprice(double newprice) { + this.newprice = newprice; + } + + /** + * 获取 + * + * @return href + */ + public String getHref() { + return href; + } + + /** + * 设置 + * + * @param href + */ + public void setHref(String href) { + this.href = href; + } + + public String toString() { + return "Book{title = " + title + ", author = " + author + ", publisher = " + publisher + ", oldprice = " + + oldprice + ", newprice = " + newprice + ", href = " + href + "}"; + } + // private String imageHref; //封面图片href地址 + +} diff --git a/src/pa/dangdang/Driver.java b/src/pa/dangdang/Driver.java new file mode 100644 index 0000000..cd1d703 --- /dev/null +++ b/src/pa/dangdang/Driver.java @@ -0,0 +1,23 @@ +package pa.dangdang; + +public class Driver { + public static void pashu() { + try { + String url = "http://bang.dangdang.com/books/bestsellers/01.00.00.00.00.00-24hours-0-0-1-1"; + Thread thread1 = new Thread(new NewsThread(url)); + thread1.start(); + + for (int i = 2; i <= 5; i++) { + + String url2 = "http://bang.dangdang.com/books/bestsellers/01.00.00.00.00.00-24hours-0-0-1-" + i; + Thread thread2 = new Thread(new NewsThread(url2)); + thread2.start(); + Thread.sleep(10000); + } + } catch (Exception e) { + + } + + } + +} diff --git a/src/pa/dangdang/NewsThread.java b/src/pa/dangdang/NewsThread.java new file mode 100644 index 0000000..0338dc4 --- /dev/null +++ b/src/pa/dangdang/NewsThread.java @@ -0,0 +1,84 @@ +package pa.dangdang; + +import java.sql.Connection; +import java.sql.PreparedStatement; + +import org.jsoup.Jsoup; +import org.jsoup.nodes.Document; +import org.jsoup.nodes.Element; +import org.jsoup.select.Elements; + +import pa.tools.CrawlerTools; +import server.tools.DBConnection; + +public class NewsThread implements Runnable { + private String urlPath; + + public NewsThread(String urlPath) { + super(); + this.urlPath = urlPath; + } + + @Override + public void run() { + + // 对爬取结果解析 + String content = CrawlerTools.get(urlPath, "GB2312"); + Document doc = Jsoup.parse(content); + Elements elements = doc.select(".bang_wrapper .bang_list_box ul li"); // 所有新闻 + Connection con = DBConnection.getConnection(); + PreparedStatement ps = null; + + String sql = null; + int index = 0; + for (Element bookelement : elements) { + index++; + if (index >= 20) + break; + // 每本图书的名称,作者,出版社,原价格、折后价格、详情url地址等信息 + String title = bookelement.select(".name a").text(); + Elements publisherInfoElements = bookelement.select(".publisher_info"); + String author = null; + String publisher = null; + if (!publisherInfoElements.isEmpty()) { + Element firstPublisherInfoElement = publisherInfoElements.get(0); + Element secondPublisherInfoElement = publisherInfoElements.get(1); + author = firstPublisherInfoElement.select("a").text(); + publisher = secondPublisherInfoElement.select("a").text(); + } else { + + } + String oprice = bookelement.select(".price_r").text(); + String nprice = bookelement.select(".price_n").first().text(); + String href = bookelement.select(".name a").attr("href"); + double oldprice = Double.parseDouble(oprice.substring(1)); + double newprice = Double.parseDouble(nprice.substring(1)); + sql = "insert into book values(?,?,?,?,?,?)";// 书名,作者名,出版社,原价格,折后价格,详情url地址 + try { + ps = con.prepareStatement(sql); + ps.setString(1, title); + ps.setString(2, author); + ps.setString(3, publisher); + ps.setDouble(4, oldprice); + ps.setDouble(5, newprice); + ps.setString(6, href); + ps.executeUpdate(); + } catch (Exception e) { + + } + System.out.println(title); + System.out.println(author); + System.out.println(publisher); + System.out.println(oldprice); + System.out.println(newprice); + System.out.println(href); + } + try { + ps.close(); + con.close(); + } catch (Exception e) { + + } + } + +} diff --git a/src/pa/tools/CrawlerTools.java b/src/pa/tools/CrawlerTools.java new file mode 100644 index 0000000..fb2da97 --- /dev/null +++ b/src/pa/tools/CrawlerTools.java @@ -0,0 +1,87 @@ +package pa.tools; + +import java.io.BufferedReader; +import java.io.IOException; +import java.io.InputStream; +import java.io.InputStreamReader; +import java.io.UnsupportedEncodingException; +import java.net.HttpURLConnection; +import java.net.URL; +import java.net.URLEncoder; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +public class CrawlerTools { + // 读取指定url的网页html字符串,需要指定网页的字符编码 + public static String get(String urlStr, String charset) { + StringBuffer buf = new StringBuffer(); + HttpURLConnection con = null; + InputStream in = null; + BufferedReader read = null; + try { + // // 待爬取的url + URL url = new URL(urlStr); + con = (HttpURLConnection) url.openConnection(); + + // 模拟浏览器发出请求,防止反爬 + con.setRequestProperty("User-Agent", + "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/101.0.4951.64 Safari/537.36 Edg/101.0.1210.53"); + int code = con.getResponseCode(); + if (code == 200) { + in = con.getInputStream(); + + read = new BufferedReader(new InputStreamReader(in, charset)); + String info = ""; + while ((info = read.readLine()) != null) { + buf.append(info); + } + } else { + System.out.println("出错:" + code); + } + + } catch (Exception e) { + e.printStackTrace(); + } finally { + if (read != null) { + try { + read.close(); + } catch (IOException e) { + e.printStackTrace(); + } + } + if (in != null) { + try { + in.close(); + } catch (IOException e) { + e.printStackTrace(); + } + } + if (con != null) { + con.disconnect(); + } + + } + + return buf.toString(); + } + + // 将url字符串里面的中文进行编码 + public static String encodingUrl(String url) { + String regex = "[\u4e00-\u9fa5]+"; + Pattern pat = Pattern.compile(regex); + Matcher mat = pat.matcher(url); + while (mat.find()) { + String hanzi = mat.group(); + System.out.println(hanzi); + String encodehanzi = ""; + try { + encodehanzi = URLEncoder.encode(hanzi, "utf-8"); + } catch (UnsupportedEncodingException e) { + e.printStackTrace(); + } + url = url.replaceAll(hanzi, encodehanzi); + } + return url; + } + +} diff --git a/src/pa/tools/HtmlParse.java b/src/pa/tools/HtmlParse.java new file mode 100644 index 0000000..12172e8 --- /dev/null +++ b/src/pa/tools/HtmlParse.java @@ -0,0 +1,54 @@ +package pa.tools; + +import java.util.ArrayList; +import java.util.regex.Matcher; +import java.util.regex.Pattern; + +//解析html格式字符串 +public class HtmlParse { + // 根据标签名称获取内容 + public static String getContent(String str, String tagName) { + String content = ""; + String regex = "<" + tagName + ".*?>(.*?)"; + Pattern pattern = Pattern.compile(regex, Pattern.CASE_INSENSITIVE); + Matcher matcher = pattern.matcher(str); + if (matcher.find()) { + content = matcher.group(1); + } + return content; + } + + // 根据标签名,属性名获取属性值 + public static String getAttributeValue(String str, String tagName, String attributeName) { + String value = ""; + // 标签与属性间至少有一个空格\\s+ + // 属性值可以用'或"括号起来['"] + // >前可以有任意个空格 + String regex = "<" + tagName + ".*?" + attributeName + "=['\"](.*?)['\"].*?>"; + Pattern pattern = Pattern.compile(regex, Pattern.CASE_INSENSITIVE); + Matcher matcher = pattern.matcher(str); + if (matcher.find()) { + value = matcher.group(1); + } + return value; + } + + // 根据标签名,class名获取内容 + public static ArrayList getContentByClassName(String str, String tagName, String className) { + ArrayList list = new ArrayList(); + String regex = "<" + tagName + ".*?class=['\"]" + className + "['\"].*?>(.*?)"; + Pattern pattern = Pattern.compile(regex, Pattern.CASE_INSENSITIVE); + Matcher matcher = pattern.matcher(str); + while (matcher.find()) { + list.add(matcher.group(1)); + } + return list; + } + + public static void main(String[] args) { + String str = " "; + String imageUrl = getAttributeValue(str, "a", "href"); + System.out.println(imageUrl); + } + +}