a
This commit is contained in:
1 parent
4776adef06
commit
b727833c17
5 files changed
+386
No files matched your search
@@ -0,0 +1,138 @@
|
||||
package pa.dangdang;
|
||||
|
||||
public class Book {
|
||||
|
||||
private String title;
|
||||
private String author; // 作者
|
||||
private String publisher; // 出版社
|
||||
private double oldprice; // 出版时间
|
||||
private double newprice; // 价格
|
||||
private String href; // 图书详情url
|
||||
|
||||
public Book() {
|
||||
}
|
||||
|
||||
public Book(String title, String author, String publisher, double oldprice, double newprice, String href) {
|
||||
this.title = title;
|
||||
this.author = author;
|
||||
this.publisher = publisher;
|
||||
this.oldprice = oldprice;
|
||||
this.newprice = newprice;
|
||||
this.href = href;
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取
|
||||
*
|
||||
* @return title
|
||||
*/
|
||||
public String getTitle() {
|
||||
return title;
|
||||
}
|
||||
|
||||
/**
|
||||
* 设置
|
||||
*
|
||||
* @param title
|
||||
*/
|
||||
public void setTitle(String title) {
|
||||
this.title = title;
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取
|
||||
*
|
||||
* @return author
|
||||
*/
|
||||
public String getAuthor() {
|
||||
return author;
|
||||
}
|
||||
|
||||
/**
|
||||
* 设置
|
||||
*
|
||||
* @param author
|
||||
*/
|
||||
public void setAuthor(String author) {
|
||||
this.author = author;
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取
|
||||
*
|
||||
* @return publisher
|
||||
*/
|
||||
public String getPublisher() {
|
||||
return publisher;
|
||||
}
|
||||
|
||||
/**
|
||||
* 设置
|
||||
*
|
||||
* @param publisher
|
||||
*/
|
||||
public void setPublisher(String publisher) {
|
||||
this.publisher = publisher;
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取
|
||||
*
|
||||
* @return oldprice
|
||||
*/
|
||||
public double getOldprice() {
|
||||
return oldprice;
|
||||
}
|
||||
|
||||
/**
|
||||
* 设置
|
||||
*
|
||||
* @param oldprice
|
||||
*/
|
||||
public void setOldprice(double oldprice) {
|
||||
this.oldprice = oldprice;
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取
|
||||
*
|
||||
* @return newprice
|
||||
*/
|
||||
public double getNewprice() {
|
||||
return newprice;
|
||||
}
|
||||
|
||||
/**
|
||||
* 设置
|
||||
*
|
||||
* @param newprice
|
||||
*/
|
||||
public void setNewprice(double newprice) {
|
||||
this.newprice = newprice;
|
||||
}
|
||||
|
||||
/**
|
||||
* 获取
|
||||
*
|
||||
* @return href
|
||||
*/
|
||||
public String getHref() {
|
||||
return href;
|
||||
}
|
||||
|
||||
/**
|
||||
* 设置
|
||||
*
|
||||
* @param href
|
||||
*/
|
||||
public void setHref(String href) {
|
||||
this.href = href;
|
||||
}
|
||||
|
||||
public String toString() {
|
||||
return "Book{title = " + title + ", author = " + author + ", publisher = " + publisher + ", oldprice = "
|
||||
+ oldprice + ", newprice = " + newprice + ", href = " + href + "}";
|
||||
}
|
||||
// private String imageHref; //封面图片href地址
|
||||
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
package pa.dangdang;
|
||||
|
||||
public class Driver {
|
||||
public static void pashu() {
|
||||
try {
|
||||
String url = "http://bang.dangdang.com/books/bestsellers/01.00.00.00.00.00-24hours-0-0-1-1";
|
||||
Thread thread1 = new Thread(new NewsThread(url));
|
||||
thread1.start();
|
||||
|
||||
for (int i = 2; i <= 5; i++) {
|
||||
|
||||
String url2 = "http://bang.dangdang.com/books/bestsellers/01.00.00.00.00.00-24hours-0-0-1-" + i;
|
||||
Thread thread2 = new Thread(new NewsThread(url2));
|
||||
thread2.start();
|
||||
Thread.sleep(10000);
|
||||
}
|
||||
} catch (Exception e) {
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
package pa.dangdang;
|
||||
|
||||
import java.sql.Connection;
|
||||
import java.sql.PreparedStatement;
|
||||
|
||||
import org.jsoup.Jsoup;
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.jsoup.nodes.Element;
|
||||
import org.jsoup.select.Elements;
|
||||
|
||||
import pa.tools.CrawlerTools;
|
||||
import server.tools.DBConnection;
|
||||
|
||||
public class NewsThread implements Runnable {
|
||||
private String urlPath;
|
||||
|
||||
public NewsThread(String urlPath) {
|
||||
super();
|
||||
this.urlPath = urlPath;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void run() {
|
||||
|
||||
// 对爬取结果解析
|
||||
String content = CrawlerTools.get(urlPath, "GB2312");
|
||||
Document doc = Jsoup.parse(content);
|
||||
Elements elements = doc.select(".bang_wrapper .bang_list_box ul li"); // 所有新闻
|
||||
Connection con = DBConnection.getConnection();
|
||||
PreparedStatement ps = null;
|
||||
|
||||
String sql = null;
|
||||
int index = 0;
|
||||
for (Element bookelement : elements) {
|
||||
index++;
|
||||
if (index >= 20)
|
||||
break;
|
||||
// 每本图书的名称,作者,出版社,原价格、折后价格、详情url地址等信息
|
||||
String title = bookelement.select(".name a").text();
|
||||
Elements publisherInfoElements = bookelement.select(".publisher_info");
|
||||
String author = null;
|
||||
String publisher = null;
|
||||
if (!publisherInfoElements.isEmpty()) {
|
||||
Element firstPublisherInfoElement = publisherInfoElements.get(0);
|
||||
Element secondPublisherInfoElement = publisherInfoElements.get(1);
|
||||
author = firstPublisherInfoElement.select("a").text();
|
||||
publisher = secondPublisherInfoElement.select("a").text();
|
||||
} else {
|
||||
|
||||
}
|
||||
String oprice = bookelement.select(".price_r").text();
|
||||
String nprice = bookelement.select(".price_n").first().text();
|
||||
String href = bookelement.select(".name a").attr("href");
|
||||
double oldprice = Double.parseDouble(oprice.substring(1));
|
||||
double newprice = Double.parseDouble(nprice.substring(1));
|
||||
sql = "insert into book values(?,?,?,?,?,?)";// 书名,作者名,出版社,原价格,折后价格,详情url地址
|
||||
try {
|
||||
ps = con.prepareStatement(sql);
|
||||
ps.setString(1, title);
|
||||
ps.setString(2, author);
|
||||
ps.setString(3, publisher);
|
||||
ps.setDouble(4, oldprice);
|
||||
ps.setDouble(5, newprice);
|
||||
ps.setString(6, href);
|
||||
ps.executeUpdate();
|
||||
} catch (Exception e) {
|
||||
|
||||
}
|
||||
System.out.println(title);
|
||||
System.out.println(author);
|
||||
System.out.println(publisher);
|
||||
System.out.println(oldprice);
|
||||
System.out.println(newprice);
|
||||
System.out.println(href);
|
||||
}
|
||||
try {
|
||||
ps.close();
|
||||
con.close();
|
||||
} catch (Exception e) {
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
package pa.tools;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.io.InputStreamReader;
|
||||
import java.io.UnsupportedEncodingException;
|
||||
import java.net.HttpURLConnection;
|
||||
import java.net.URL;
|
||||
import java.net.URLEncoder;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
public class CrawlerTools {
|
||||
// 读取指定url的网页html字符串,需要指定网页的字符编码
|
||||
public static String get(String urlStr, String charset) {
|
||||
StringBuffer buf = new StringBuffer();
|
||||
HttpURLConnection con = null;
|
||||
InputStream in = null;
|
||||
BufferedReader read = null;
|
||||
try {
|
||||
// // 待爬取的url
|
||||
URL url = new URL(urlStr);
|
||||
con = (HttpURLConnection) url.openConnection();
|
||||
|
||||
// 模拟浏览器发出请求,防止反爬
|
||||
con.setRequestProperty("User-Agent",
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/101.0.4951.64 Safari/537.36 Edg/101.0.1210.53");
|
||||
int code = con.getResponseCode();
|
||||
if (code == 200) {
|
||||
in = con.getInputStream();
|
||||
|
||||
read = new BufferedReader(new InputStreamReader(in, charset));
|
||||
String info = "";
|
||||
while ((info = read.readLine()) != null) {
|
||||
buf.append(info);
|
||||
}
|
||||
} else {
|
||||
System.out.println("出错:" + code);
|
||||
}
|
||||
|
||||
} catch (Exception e) {
|
||||
e.printStackTrace();
|
||||
} finally {
|
||||
if (read != null) {
|
||||
try {
|
||||
read.close();
|
||||
} catch (IOException e) {
|
||||
e.printStackTrace();
|
||||
}
|
||||
}
|
||||
if (in != null) {
|
||||
try {
|
||||
in.close();
|
||||
} catch (IOException e) {
|
||||
e.printStackTrace();
|
||||
}
|
||||
}
|
||||
if (con != null) {
|
||||
con.disconnect();
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
return buf.toString();
|
||||
}
|
||||
|
||||
// 将url字符串里面的中文进行编码
|
||||
public static String encodingUrl(String url) {
|
||||
String regex = "[\u4e00-\u9fa5]+";
|
||||
Pattern pat = Pattern.compile(regex);
|
||||
Matcher mat = pat.matcher(url);
|
||||
while (mat.find()) {
|
||||
String hanzi = mat.group();
|
||||
System.out.println(hanzi);
|
||||
String encodehanzi = "";
|
||||
try {
|
||||
encodehanzi = URLEncoder.encode(hanzi, "utf-8");
|
||||
} catch (UnsupportedEncodingException e) {
|
||||
e.printStackTrace();
|
||||
}
|
||||
url = url.replaceAll(hanzi, encodehanzi);
|
||||
}
|
||||
return url;
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
package pa.tools;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
//解析html格式字符串
|
||||
public class HtmlParse {
|
||||
// 根据标签名称获取内容
|
||||
public static String getContent(String str, String tagName) {
|
||||
String content = "";
|
||||
String regex = "<" + tagName + ".*?>(.*?)</" + tagName + ">";
|
||||
Pattern pattern = Pattern.compile(regex, Pattern.CASE_INSENSITIVE);
|
||||
Matcher matcher = pattern.matcher(str);
|
||||
if (matcher.find()) {
|
||||
content = matcher.group(1);
|
||||
}
|
||||
return content;
|
||||
}
|
||||
|
||||
// 根据标签名,属性名获取属性值
|
||||
public static String getAttributeValue(String str, String tagName, String attributeName) {
|
||||
String value = "";
|
||||
// 标签与属性间至少有一个空格\\s+
|
||||
// 属性值可以用'或"括号起来['"]
|
||||
// >前可以有任意个空格
|
||||
String regex = "<" + tagName + ".*?" + attributeName + "=['\"](.*?)['\"].*?>";
|
||||
Pattern pattern = Pattern.compile(regex, Pattern.CASE_INSENSITIVE);
|
||||
Matcher matcher = pattern.matcher(str);
|
||||
if (matcher.find()) {
|
||||
value = matcher.group(1);
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
// 根据标签名,class名获取内容
|
||||
public static ArrayList<String> getContentByClassName(String str, String tagName, String className) {
|
||||
ArrayList<String> list = new ArrayList<String>();
|
||||
String regex = "<" + tagName + ".*?class=['\"]" + className + "['\"].*?>(.*?)</" + tagName + ">";
|
||||
Pattern pattern = Pattern.compile(regex, Pattern.CASE_INSENSITIVE);
|
||||
Matcher matcher = pattern.matcher(str);
|
||||
while (matcher.find()) {
|
||||
list.add(matcher.group(1));
|
||||
}
|
||||
return list;
|
||||
}
|
||||
|
||||
public static void main(String[] args) {
|
||||
String str = "<a class=\"nbg\" href=\"https://book.douban.com/subject/27104286/\" onclick=\"moreurl(this,{i:'0',query:'',subject_id:'27104286',from:'book_subject_search'})\"> <img class=\"\" src=\"https://img3.doubanio.com/view/subject/s/public/s29508790.jpg\" width=\"90\"> </a> ";
|
||||
String imageUrl = getAttributeValue(str, "a", "href");
|
||||
System.out.println(imageUrl);
|
||||
}
|
||||
|
||||
}
|
||||
Reference in new issue
Block a user