From c79044416e4bd76519d4e2e0bbf58c7164089ee3 Mon Sep 17 00:00:00 2001
From: wrzzx <863076034@qq.com>
Date: Sun, 19 Feb 2017 14:23:50 +0800
Subject: [PATCH] put the items which has failed to dowanload homepage-page
into db(openhub)
---
.../fetch_networks/.idea/workspace.xml | 496 +++++++++-
new_osseanextractor/.idea/workspace.xml | 926 +++++++++++-------
new_osseanextractor/osseanextractor.iml | 1 -
new_osseanextractor/pom.xml | 1 -
.../net/trustie/dao/OpenHubRetry_Dao.java | 46 +
.../java/net/trustie/model/openhub_Model.java | 56 +-
.../trustie/model/openhub_retry_Model.java | 36 +
.../net/trustie/model/sourceforge_Model.java | 6 +-
.../java/net/trustie/one/ExtractThread.java | 12 +-
.../net/trustie/one/OpenhubReExtractor.java | 2 +-
.../utils/ExtractMutilLink4Openhub.java | 13 +-
11 files changed, 1172 insertions(+), 423 deletions(-)
create mode 100644 new_osseanextractor/src/main/java/net/trustie/dao/OpenHubRetry_Dao.java
create mode 100644 new_osseanextractor/src/main/java/net/trustie/model/openhub_retry_Model.java
diff --git a/crawler/moreSmarterCrawler/fetch_networks/.idea/workspace.xml b/crawler/moreSmarterCrawler/fetch_networks/.idea/workspace.xml
index 0518fc684..2833dd458 100644
--- a/crawler/moreSmarterCrawler/fetch_networks/.idea/workspace.xml
+++ b/crawler/moreSmarterCrawler/fetch_networks/.idea/workspace.xml
@@ -4,9 +4,13 @@
-
+
+
+
+
+
@@ -33,28 +37,48 @@
-
-
+
+
-
-
+
+
-
-
+
+
-
-
+
+
+
+
+
+
+
+
+
+
+
+
-
+
+
+
+
+
+
+
+
+
+
+
@@ -144,7 +168,6 @@
-
@@ -157,10 +180,241 @@
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
@@ -624,11 +878,14 @@
+
+
+
-
+
@@ -641,12 +898,11 @@
-
+
-
@@ -659,7 +915,7 @@
-
+
@@ -671,6 +927,7 @@
+
@@ -694,7 +951,9 @@
-
+
+
+
@@ -702,9 +961,7 @@
-
-
-
+
@@ -712,7 +969,9 @@
-
+
+
+
@@ -720,8 +979,70 @@
+
+
+
+
+
+
+
+
-
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
@@ -730,7 +1051,9 @@
-
+
+
+
@@ -738,7 +1061,9 @@
-
+
+
+
@@ -746,7 +1071,9 @@
-
+
+
+
@@ -754,7 +1081,9 @@
-
+
+
+
@@ -762,15 +1091,9 @@
-
-
-
-
-
-
-
-
-
+
+
+
@@ -786,11 +1109,7 @@
-
-
-
-
-
+
@@ -802,18 +1121,6 @@
-
-
-
-
-
-
-
-
-
-
-
-
@@ -826,19 +1133,100 @@
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
-
-
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
-
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/new_osseanextractor/.idea/workspace.xml b/new_osseanextractor/.idea/workspace.xml
index cf7385e54..665ec2828 100644
--- a/new_osseanextractor/.idea/workspace.xml
+++ b/new_osseanextractor/.idea/workspace.xml
@@ -22,70 +22,75 @@
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
+
-
-
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
gethome
gethomepage
readSitesFromConfig
- valid
tags
description
抽取时间
抽取
+ errorpage
+ after
+ lastID
+ getPages
+ valid
+ getValidHomepage
+ validate
@@ -100,13 +105,21 @@
-
+
+
+
+
-
+
+
+
+
+
+
@@ -219,8 +232,9 @@
-
+
+
@@ -272,6 +286,104 @@
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
@@ -303,6 +415,36 @@
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
@@ -372,6 +514,10 @@
+
+
+
+
@@ -393,7 +539,6 @@
-
@@ -419,6 +564,8 @@
+
+
@@ -906,48 +1053,59 @@
-
+
+
+
-
+
+
+
+
+
+
+
+
+
+
-
-
-
-
-
-
-
-
-
+
+
+
+
+
+
+
+
+
+
+
+
-
-
+
-
-
@@ -1001,25 +1159,13 @@
- file://$PROJECT_DIR$/src/main/java/net/trustie/one/OpenhubReExtractor.java
- 126
+ file://$PROJECT_DIR$/src/main/java/net/trustie/model/openhub_Model.java
+ 1306
-
-
-
- file://$PROJECT_DIR$/src/main/java/net/trustie/one/OpenhubReExtractor.java
- 94
-
-
-
-
- file://$PROJECT_DIR$/src/main/java/net/trustie/one/ExtractThread.java
- 100
-
-
+
-
+
@@ -1028,67 +1174,15 @@
-
+
-
-
-
-
-
-
-
-
-
-
-
-
+
+
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
+
@@ -1096,131 +1190,6 @@
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
@@ -1229,53 +1198,10 @@
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
+
+
@@ -1284,80 +1210,392 @@
-
-
-
-
-
-
-
-
-
-
-
-
-
-
-
+
-
-
-
-
-
-
-
-
-
+
+
-
+
-
-
+
+
-
+
-
-
+
+
-
+
-
-
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
-
-
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
-
-
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/new_osseanextractor/osseanextractor.iml b/new_osseanextractor/osseanextractor.iml
index dbab3e2c4..1bbfc8624 100644
--- a/new_osseanextractor/osseanextractor.iml
+++ b/new_osseanextractor/osseanextractor.iml
@@ -11,7 +11,6 @@
-
diff --git a/new_osseanextractor/pom.xml b/new_osseanextractor/pom.xml
index dd0db06de..eed203f34 100644
--- a/new_osseanextractor/pom.xml
+++ b/new_osseanextractor/pom.xml
@@ -82,6 +82,5 @@
commons-io
2.4
-
diff --git a/new_osseanextractor/src/main/java/net/trustie/dao/OpenHubRetry_Dao.java b/new_osseanextractor/src/main/java/net/trustie/dao/OpenHubRetry_Dao.java
new file mode 100644
index 000000000..98b4aed65
--- /dev/null
+++ b/new_osseanextractor/src/main/java/net/trustie/dao/OpenHubRetry_Dao.java
@@ -0,0 +1,46 @@
+package net.trustie.dao;
+
+import net.trustie.model.openhub_retry_Model;
+
+import java.sql.Connection;
+import java.sql.DriverManager;
+import java.sql.PreparedStatement;
+import java.sql.ResultSet;
+import java.sql.SQLException;
+import java.sql.Statement;
+
+/**
+ * Created by zaihuilvcha on 2017/2/18.
+ */
+public class OpenHubRetry_Dao {
+ private Connection conn = null;
+ private Statement stmt = null;
+
+ public OpenHubRetry_Dao() {
+ try {
+ Class.forName("com.mysql.jdbc.Driver");
+ String url = "jdbc:mysql://localhost:3306/extract_result?user=root&password=123456";
+ conn = DriverManager.getConnection(url);
+ stmt = conn.createStatement();
+ } catch (ClassNotFoundException e) {
+ e.printStackTrace();
+ } catch (SQLException e) {
+ e.printStackTrace();
+ }
+
+ }
+
+ public int add(openhub_retry_Model oprm) {
+ try {
+ String sql = "INSERT INTO `extract_result`.`openhub_download_fail` (`url`, `html`) VALUES (?, ?);";
+ PreparedStatement ps = conn.prepareStatement(sql);
+ ps.setString(1, oprm.getUrl());
+ ps.setString(2,oprm.getHtml());
+ return ps.executeUpdate();
+ } catch (SQLException e) {
+ e.printStackTrace();
+ }
+ return -1;
+ }
+
+}
diff --git a/new_osseanextractor/src/main/java/net/trustie/model/openhub_Model.java b/new_osseanextractor/src/main/java/net/trustie/model/openhub_Model.java
index 0b015f416..7c1935102 100644
--- a/new_osseanextractor/src/main/java/net/trustie/model/openhub_Model.java
+++ b/new_osseanextractor/src/main/java/net/trustie/model/openhub_Model.java
@@ -6,7 +6,8 @@ import java.util.ArrayList;
import java.util.Date;
import java.util.List;
-import net.trustie.utils.DateHandler;
+import core.*;
+import net.trustie.dao.OpenHubRetry_Dao;
import net.trustie.utils.ExtractMutilLink4Openhub;
import net.trustie.utils.Seperator;
import net.trustie.utils.StringHandler;
@@ -18,9 +19,8 @@ import org.jsoup.nodes.Document;
import org.jsoup.nodes.Element;
import org.jsoup.select.Elements;
-import core.AfterExtractor;
-import core.Page;
-import core.ValidateExtractor;
+import org.springframework.context.ApplicationContext;
+import org.springframework.context.support.ClassPathXmlApplicationContext;
import us.codecraft.webmagic.model.annotation.ExtractBy;
@ExtractBy("//div[@id='projects_show_page']")
@@ -49,10 +49,11 @@ public class openhub_Model implements AfterExtractor, ValidateExtractor {
// "| //*[@id='projects_show_page']/div[2]/div[3]/div[2]/div/*/*/a/regex(\".*Homepage\",1)/@href" //jquery multi links
// +"| //*[@id='projects_show_page']/div[2]/div[4]/div[2]/div/*/*/a/regex(\".*Homepage\",1)/@href "
//)
- @ExtractBy(value="//a/regex(\"\",1) " +
- " | //a/regex(\"\",1)" +
- " | //a/regex(\"\",1)")
-private List homepages = new ArrayList();
+// @ExtractBy(value="//a/regex(\"\",1) " +
+// " | //a/regex(\"\",1)" +
+// " | //a/regex(\"\",1)")
+ @ExtractBy(value="//a/regex(\"\",1) ")
+ private List homepages = new ArrayList();
private static String homepage ="";
///////////////////////////////////
@@ -401,6 +402,7 @@ private List homepages = new ArrayList();
}
+
@Override
public void validate(Page page) {
//
@@ -423,18 +425,43 @@ private List homepages = new ArrayList();
}
this.rateLevel = this.rateLevel.substring(0, this.rateLevel.indexOf("/"));
+
+
if (StringHandler.isAtLeastOneBlank(this.name, this.activity
,this.description
- ,getHomepage()
+ ,this.getHomepage()
/* ,
* this.licenses
*/)) {
page.setResultSkip(this, true);
+
+ //多homepage抽取时,下载失败,需要将对应page存库.downloadFailFlag为存库失败标志
+ if(downloadFailFlag) {
+
+ //存库开始,先将下载失败标志置为假
+ downloadFailFlag = false;
+
+ System.out.println("homepage页面下载失败,条目准备入库...");
+
+ openhub_retry_Model oprm = new openhub_retry_Model();
+ oprm.setUrl(page.getPageUrl());
+ oprm.setHtml(page.getRawText());
+
+ OpenHubRetry_Dao opDao = new OpenHubRetry_Dao();
+ opDao.add(oprm);
+
+ System.out.println("下载失败条目已存入数据库...");
+
+ }
+
return;
}
+
+
+
}
private void handleQuickRef(Element quickRef) {
@@ -1274,6 +1301,7 @@ private List homepages = new ArrayList();
this.history = history;
}
+ private boolean downloadFailFlag = false;
public String getValidHomepage(String homePage){
String result = homePage;
@@ -1281,6 +1309,16 @@ private List homepages = new ArrayList();
if(!homePage.equals("") && homePage.startsWith("/")){
//获取Homepage列表中的Homepage
result = new ExtractMutilLink4Openhub().extractLinks(homePage);
+ /**
+ * result=""时,需要对相应page加入存库处理。
+ * 这里必然是多homepage情况。所以此处若正常则result不为空,若为空则说明下载失败了,必须存库
+ */
+ if(result.equals("")) {
+
+ //存库标志位置为true
+ downloadFailFlag = true;
+
+ }
}
return result;
}
diff --git a/new_osseanextractor/src/main/java/net/trustie/model/openhub_retry_Model.java b/new_osseanextractor/src/main/java/net/trustie/model/openhub_retry_Model.java
new file mode 100644
index 000000000..892e742a3
--- /dev/null
+++ b/new_osseanextractor/src/main/java/net/trustie/model/openhub_retry_Model.java
@@ -0,0 +1,36 @@
+package net.trustie.model;
+
+/**
+ * Created by zaihuilvcha on 2017/2/18.
+ */
+public class openhub_retry_Model {
+ String url;
+ String html;
+
+
+
+ public void setUrl(String url) {
+ this.url = url;
+ }
+
+ public void setHtml(String html) {
+ this.html = html;
+ }
+
+ public String getUrl() {
+
+ return url;
+ }
+
+ public String getHtml() {
+ return html;
+ }
+
+ @Override
+ public String toString() {
+ return "openhub_retry_Model{" +
+ "url='" + url + '\'' +
+ ", html='" + html + '\'' +
+ '}';
+ }
+}
diff --git a/new_osseanextractor/src/main/java/net/trustie/model/sourceforge_Model.java b/new_osseanextractor/src/main/java/net/trustie/model/sourceforge_Model.java
index e740c11b8..0adb501e0 100644
--- a/new_osseanextractor/src/main/java/net/trustie/model/sourceforge_Model.java
+++ b/new_osseanextractor/src/main/java/net/trustie/model/sourceforge_Model.java
@@ -28,9 +28,9 @@ import java.util.regex.Pattern;
public class sourceforge_Model implements AfterExtractor, ValidateExtractor{
///////////////
-
-@ExtractBy("//a[@id='homepage']/@href " +
- " | //*[@class='homepage-link']/a/@href")
+//@ExtractBy("//a[@id='homepage']/@href " +
+// " | //*[@class='homepage-link']/a/@href")
+@ExtractBy("//*[@id='homepage']/@href ")
private String homepage;
//////////////////
diff --git a/new_osseanextractor/src/main/java/net/trustie/one/ExtractThread.java b/new_osseanextractor/src/main/java/net/trustie/one/ExtractThread.java
index 028d9b500..8468af9a3 100644
--- a/new_osseanextractor/src/main/java/net/trustie/one/ExtractThread.java
+++ b/new_osseanextractor/src/main/java/net/trustie/one/ExtractThread.java
@@ -84,13 +84,16 @@ public class ExtractThread implements Runnable{
Extractor extractor = new Extractor();
RawPage result = null;
while(pages.size() > 0){
-
+ /**
+ * 注意这里的page就是detail表中的一条记录。
+ */
for( RawPage page : pages){
try{
long startTime=System.currentTimeMillis(); //获取开始时间
result = extractor.extract(page,pageModel);
long endTime=System.currentTimeMillis(); //获取结束时间
-// System.out.println("页面抽取时间: "+(endTime-startTime)+"ms");
+// System.out.println("页面抽取时间: "+(endTime-startTime)+"ms");
+
startTime=System.currentTimeMillis(); //获取开始时间
//持久化 并 更新抽取历史
saveResult(site,result);
@@ -98,10 +101,11 @@ public class ExtractThread implements Runnable{
// System.out.println("结果保存时间: "+(endTime-startTime)+"ms");
}catch (Exception e){
- e.printStackTrace();
-// pageErrorOutPut.returnErrorPage(page, e);错误页面
+// e.printStackTrace();
+// pageErrorOutPut.returnErrorPage(page, e); //错误页面
}
}
+ //结果入库之后再更新lastId.注意,处理一批才会存一次
updateLastId(site,lastId + pages.size());
lastId = getLastId(site);
pages = getPages(site,lastId);
diff --git a/new_osseanextractor/src/main/java/net/trustie/one/OpenhubReExtractor.java b/new_osseanextractor/src/main/java/net/trustie/one/OpenhubReExtractor.java
index a88c412fe..ed9fe8e25 100644
--- a/new_osseanextractor/src/main/java/net/trustie/one/OpenhubReExtractor.java
+++ b/new_osseanextractor/src/main/java/net/trustie/one/OpenhubReExtractor.java
@@ -95,7 +95,7 @@ class ReExtractThread implements Runnable{
saveResult(site, rawPage);
}catch (Exception e){
e.printStackTrace();
- // pageErrorOutPut.returnErrorPage(page, e);错误页面
+// pageErrorOutPut.returnErrorPage(page, e); //错误页面
}
}
//更新抽取游标
diff --git a/new_osseanextractor/src/main/java/net/trustie/utils/ExtractMutilLink4Openhub.java b/new_osseanextractor/src/main/java/net/trustie/utils/ExtractMutilLink4Openhub.java
index a7e0d51cf..99d9b02ef 100644
--- a/new_osseanextractor/src/main/java/net/trustie/utils/ExtractMutilLink4Openhub.java
+++ b/new_osseanextractor/src/main/java/net/trustie/utils/ExtractMutilLink4Openhub.java
@@ -12,7 +12,7 @@ import java.util.List;
* Description:抽取openhub中多链接的项目
*/
public class ExtractMutilLink4Openhub implements PageProcessor{
- private Site site = Site.me().setRetryTimes(10).setSleepTime(500);
+ private Site site = Site.me().setRetryTimes(5).setSleepTime(500);
private static List homepages;
private static Page _page;
@@ -40,27 +40,28 @@ public class ExtractMutilLink4Openhub implements PageProcessor{
}
}catch(Exception e) {
+
}
return ExtractMutilLink4Openhub.homepages;
}
/**
- * 获取多个link,用“;”间隔
+ * 获取多个link,用“;”间隔.此方法为二次抽取的入口方法
* @param url
* @return
*/
public String extractLinks(String url){
String homepage = "";
- doExtract(url);
+ doExtract(url); //(如果成功了的话)得到了homepages
try {
- List tempPages = getHomepages(url);
+ List tempPages = getHomepages(url); //以防初次抽取的homepages没抽成功,若没成功的话通过getHomepages继续抽
if (tempPages != null && !tempPages.isEmpty()) {
for (String link : tempPages) {
if(!homepage.contains(link)){//避免加入重复的链接
homepage += link + ";";
}
}
- homepage = homepage.substring(0, homepage.lastIndexOf(";"));
+ homepage = homepage.substring(0, homepage.lastIndexOf(";")); //将结果最后的“;”去除
}
}catch (Exception e){
e.printStackTrace();
@@ -89,7 +90,7 @@ public class ExtractMutilLink4Openhub implements PageProcessor{
* 抽取Homepage核心方法
* @param url 抽取目标url链接
*/
- public static void doExtract(String url){
+ public static void doExtract(String url) {
Spider.create(extractMutilLink4Openhub).addUrl(new String("https://www.openhub.net"+url)).thread(1).run();
}