This commit is contained in:
xixu-me committed 2024-07-01 22:25:24 +08:00
1 parent 4776adef06
commit b727833c17
5 files changed
+386

No files matched your search

+138
View File
@@ -0,0 +1,138 @@
package pa.dangdang;
public class Book {
private String title;
private String author; // 作者
private String publisher; // 出版社
private double oldprice; // 出版时间
private double newprice; // 价格
private String href; // 图书详情url
public Book() {
}
public Book(String title, String author, String publisher, double oldprice, double newprice, String href) {
this.title = title;
this.author = author;
this.publisher = publisher;
this.oldprice = oldprice;
this.newprice = newprice;
this.href = href;
}
/**
* 获取
*
* @return title
*/
public String getTitle() {
return title;
}
/**
* 设置
*
* @param title
*/
public void setTitle(String title) {
this.title = title;
}
/**
* 获取
*
* @return author
*/
public String getAuthor() {
return author;
}
/**
* 设置
*
* @param author
*/
public void setAuthor(String author) {
this.author = author;
}
/**
* 获取
*
* @return publisher
*/
public String getPublisher() {
return publisher;
}
/**
* 设置
*
* @param publisher
*/
public void setPublisher(String publisher) {
this.publisher = publisher;
}
/**
* 获取
*
* @return oldprice
*/
public double getOldprice() {
return oldprice;
}
/**
* 设置
*
* @param oldprice
*/
public void setOldprice(double oldprice) {
this.oldprice = oldprice;
}
/**
* 获取
*
* @return newprice
*/
public double getNewprice() {
return newprice;
}
/**
* 设置
*
* @param newprice
*/
public void setNewprice(double newprice) {
this.newprice = newprice;
}
/**
* 获取
*
* @return href
*/
public String getHref() {
return href;
}
/**
* 设置
*
* @param href
*/
public void setHref(String href) {
this.href = href;
}
public String toString() {
return "Book{title = " + title + ", author = " + author + ", publisher = " + publisher + ", oldprice = "
+ oldprice + ", newprice = " + newprice + ", href = " + href + "}";
}
// private String imageHref; //封面图片href地址
}
+23
View File
@@ -0,0 +1,23 @@
package pa.dangdang;
public class Driver {
public static void pashu() {
try {
String url = "http://bang.dangdang.com/books/bestsellers/01.00.00.00.00.00-24hours-0-0-1-1";
Thread thread1 = new Thread(new NewsThread(url));
thread1.start();
for (int i = 2; i <= 5; i++) {
String url2 = "http://bang.dangdang.com/books/bestsellers/01.00.00.00.00.00-24hours-0-0-1-" + i;
Thread thread2 = new Thread(new NewsThread(url2));
thread2.start();
Thread.sleep(10000);
}
} catch (Exception e) {
}
}
}
+84
View File
@@ -0,0 +1,84 @@
package pa.dangdang;
import java.sql.Connection;
import java.sql.PreparedStatement;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.jsoup.nodes.Element;
import org.jsoup.select.Elements;
import pa.tools.CrawlerTools;
import server.tools.DBConnection;
public class NewsThread implements Runnable {
private String urlPath;
public NewsThread(String urlPath) {
super();
this.urlPath = urlPath;
}
@Override
public void run() {
// 对爬取结果解析
String content = CrawlerTools.get(urlPath, "GB2312");
Document doc = Jsoup.parse(content);
Elements elements = doc.select(".bang_wrapper .bang_list_box ul li"); // 所有新闻
Connection con = DBConnection.getConnection();
PreparedStatement ps = null;
String sql = null;
int index = 0;
for (Element bookelement : elements) {
index++;
if (index >= 20)
break;
// 每本图书的名称,作者,出版社,原价格、折后价格、详情url地址等信息
String title = bookelement.select(".name a").text();
Elements publisherInfoElements = bookelement.select(".publisher_info");
String author = null;
String publisher = null;
if (!publisherInfoElements.isEmpty()) {
Element firstPublisherInfoElement = publisherInfoElements.get(0);
Element secondPublisherInfoElement = publisherInfoElements.get(1);
author = firstPublisherInfoElement.select("a").text();
publisher = secondPublisherInfoElement.select("a").text();
} else {
}
String oprice = bookelement.select(".price_r").text();
String nprice = bookelement.select(".price_n").first().text();
String href = bookelement.select(".name a").attr("href");
double oldprice = Double.parseDouble(oprice.substring(1));
double newprice = Double.parseDouble(nprice.substring(1));
sql = "insert into book values(?,?,?,?,?,?)";// 书名,作者名,出版社,原价格,折后价格,详情url地址
try {
ps = con.prepareStatement(sql);
ps.setString(1, title);
ps.setString(2, author);
ps.setString(3, publisher);
ps.setDouble(4, oldprice);
ps.setDouble(5, newprice);
ps.setString(6, href);
ps.executeUpdate();
} catch (Exception e) {
}
System.out.println(title);
System.out.println(author);
System.out.println(publisher);
System.out.println(oldprice);
System.out.println(newprice);
System.out.println(href);
}
try {
ps.close();
con.close();
} catch (Exception e) {
}
}
}
+87
View File
@@ -0,0 +1,87 @@
package pa.tools;
import java.io.BufferedReader;
import java.io.IOException;
import java.io.InputStream;
import java.io.InputStreamReader;
import java.io.UnsupportedEncodingException;
import java.net.HttpURLConnection;
import java.net.URL;
import java.net.URLEncoder;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
public class CrawlerTools {
// 读取指定url的网页html字符串,需要指定网页的字符编码
public static String get(String urlStr, String charset) {
StringBuffer buf = new StringBuffer();
HttpURLConnection con = null;
InputStream in = null;
BufferedReader read = null;
try {
// // 待爬取的url
URL url = new URL(urlStr);
con = (HttpURLConnection) url.openConnection();
// 模拟浏览器发出请求,防止反爬
con.setRequestProperty("User-Agent",
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/101.0.4951.64 Safari/537.36 Edg/101.0.1210.53");
int code = con.getResponseCode();
if (code == 200) {
in = con.getInputStream();
read = new BufferedReader(new InputStreamReader(in, charset));
String info = "";
while ((info = read.readLine()) != null) {
buf.append(info);
}
} else {
System.out.println("出错:" + code);
}
} catch (Exception e) {
e.printStackTrace();
} finally {
if (read != null) {
try {
read.close();
} catch (IOException e) {
e.printStackTrace();
}
}
if (in != null) {
try {
in.close();
} catch (IOException e) {
e.printStackTrace();
}
}
if (con != null) {
con.disconnect();
}
}
return buf.toString();
}
// 将url字符串里面的中文进行编码
public static String encodingUrl(String url) {
String regex = "[\u4e00-\u9fa5]+";
Pattern pat = Pattern.compile(regex);
Matcher mat = pat.matcher(url);
while (mat.find()) {
String hanzi = mat.group();
System.out.println(hanzi);
String encodehanzi = "";
try {
encodehanzi = URLEncoder.encode(hanzi, "utf-8");
} catch (UnsupportedEncodingException e) {
e.printStackTrace();
}
url = url.replaceAll(hanzi, encodehanzi);
}
return url;
}
}
+54
View File
@@ -0,0 +1,54 @@
package pa.tools;
import java.util.ArrayList;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
//解析html格式字符串
public class HtmlParse {
// 根据标签名称获取内容
public static String getContent(String str, String tagName) {
String content = "";
String regex = "<" + tagName + ".*?>(.*?)</" + tagName + ">";
Pattern pattern = Pattern.compile(regex, Pattern.CASE_INSENSITIVE);
Matcher matcher = pattern.matcher(str);
if (matcher.find()) {
content = matcher.group(1);
}
return content;
}
// 根据标签名,属性名获取属性值
public static String getAttributeValue(String str, String tagName, String attributeName) {
String value = "";
// 标签与属性间至少有一个空格\\s+
// 属性值可以用'或"括号起来['"]
// >前可以有任意个空格
String regex = "<" + tagName + ".*?" + attributeName + "=['\"](.*?)['\"].*?>";
Pattern pattern = Pattern.compile(regex, Pattern.CASE_INSENSITIVE);
Matcher matcher = pattern.matcher(str);
if (matcher.find()) {
value = matcher.group(1);
}
return value;
}
// 根据标签名,class名获取内容
public static ArrayList<String> getContentByClassName(String str, String tagName, String className) {
ArrayList<String> list = new ArrayList<String>();
String regex = "<" + tagName + ".*?class=['\"]" + className + "['\"].*?>(.*?)</" + tagName + ">";
Pattern pattern = Pattern.compile(regex, Pattern.CASE_INSENSITIVE);
Matcher matcher = pattern.matcher(str);
while (matcher.find()) {
list.add(matcher.group(1));
}
return list;
}
public static void main(String[] args) {
String str = "<a class=\"nbg\" href=\"https://book.douban.com/subject/27104286/\" onclick=\"moreurl(this,{i:'0',query:'',subject_id:'27104286',from:'book_subject_search'})\"> <img class=\"\" src=\"https://img3.doubanio.com/view/subject/s/public/s29508790.jpg\" width=\"90\"> </a> ";
String imageUrl = getAttributeValue(str, "a", "href");
System.out.println(imageUrl);
}
}