upadte ossean web-front
This commit is contained in:
commit
a74a0ea939
Binary file not shown.
|
|
@ -0,0 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="CompilerConfiguration">
|
||||
<bytecodeTargetLevel target="1.8" />
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,9 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<module type="JAVA_MODULE" version="4">
|
||||
<component name="NewModuleRootManager" inherit-compiler-output="true">
|
||||
<exclude-output />
|
||||
<content url="file://$MODULE_DIR$" />
|
||||
<orderEntry type="inheritedJdk" />
|
||||
<orderEntry type="sourceFolder" forTests="false" />
|
||||
</component>
|
||||
</module>
|
||||
|
|
@ -0,0 +1,71 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="MavenImportPreferences">
|
||||
<option name="generalSettings">
|
||||
<MavenGeneralSettings>
|
||||
<option name="mavenHome" value="/usr/local/Cellar/maven/3.5.0/libexec" />
|
||||
</MavenGeneralSettings>
|
||||
</option>
|
||||
</component>
|
||||
<component name="masterDetails">
|
||||
<states>
|
||||
<state key="GlobalLibrariesConfigurable.UI">
|
||||
<settings>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
<state key="JdkListConfigurable.UI">
|
||||
<settings>
|
||||
<last-edited>1.8</last-edited>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
<state key="ProjectJDKs.UI">
|
||||
<settings>
|
||||
<last-edited>1.8</last-edited>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
<state key="ProjectLibrariesConfigurable.UI">
|
||||
<settings>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
<state key="ScopeChooserConfigurable.UI">
|
||||
<settings>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
</states>
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,8 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="ProjectModuleManager">
|
||||
<modules>
|
||||
<module fileurl="file://$PROJECT_DIR$/.idea/crawler.iml" filepath="$PROJECT_DIR$/.idea/crawler.iml" />
|
||||
</modules>
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,251 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="ChangeListManager">
|
||||
<list default="true" id="32a73eed-3716-4f88-8891-2b38051ca637" name="Default" comment="" />
|
||||
<option name="EXCLUDED_CONVERTED_TO_IGNORED" value="true" />
|
||||
<option name="TRACKING_ENABLED" value="true" />
|
||||
<option name="SHOW_DIALOG" value="false" />
|
||||
<option name="HIGHLIGHT_CONFLICTS" value="true" />
|
||||
<option name="HIGHLIGHT_NON_ACTIVE_CHANGELIST" value="false" />
|
||||
<option name="LAST_RESOLUTION" value="IGNORE" />
|
||||
</component>
|
||||
<component name="ExecutionTargetManager" SELECTED_TARGET="default_target" />
|
||||
<component name="GradleLocalSettings">
|
||||
<option name="externalProjectsViewState">
|
||||
<projects_view />
|
||||
</option>
|
||||
</component>
|
||||
<component name="JsBuildToolGruntFileManager" detection-done="true" sorting="DEFINITION_ORDER" />
|
||||
<component name="JsBuildToolPackageJson" detection-done="true" sorting="DEFINITION_ORDER" />
|
||||
<component name="JsGulpfileManager">
|
||||
<detection-done>true</detection-done>
|
||||
<sorting>DEFINITION_ORDER</sorting>
|
||||
</component>
|
||||
<component name="ProjectFrameBounds">
|
||||
<option name="width" value="1280" />
|
||||
<option name="height" value="800" />
|
||||
</component>
|
||||
<component name="ProjectView">
|
||||
<navigator currentView="ProjectPane" proportions="" version="1">
|
||||
<flattenPackages />
|
||||
<showMembers />
|
||||
<showModules />
|
||||
<showLibraryContents />
|
||||
<hideEmptyPackages />
|
||||
<abbreviatePackageNames />
|
||||
<autoscrollToSource />
|
||||
<autoscrollFromSource />
|
||||
<sortByType />
|
||||
<manualOrder />
|
||||
<foldersAlwaysOnTop value="true" />
|
||||
</navigator>
|
||||
<panes>
|
||||
<pane id="ProjectPane">
|
||||
<subPane>
|
||||
<PATH>
|
||||
<PATH_ELEMENT>
|
||||
<option name="myItemId" value="crawler" />
|
||||
<option name="myItemType" value="com.intellij.ide.projectView.impl.nodes.ProjectViewProjectNode" />
|
||||
</PATH_ELEMENT>
|
||||
<PATH_ELEMENT>
|
||||
<option name="myItemId" value="crawler" />
|
||||
<option name="myItemType" value="com.intellij.ide.projectView.impl.nodes.PsiDirectoryNode" />
|
||||
</PATH_ELEMENT>
|
||||
<PATH_ELEMENT>
|
||||
<option name="myItemId" value="moreSmarterCrawler" />
|
||||
<option name="myItemType" value="com.intellij.ide.projectView.impl.nodes.PsiDirectoryNode" />
|
||||
</PATH_ELEMENT>
|
||||
<PATH_ELEMENT>
|
||||
<option name="myItemId" value="fetch_networks" />
|
||||
<option name="myItemType" value="com.intellij.ide.projectView.impl.nodes.PsiDirectoryNode" />
|
||||
</PATH_ELEMENT>
|
||||
</PATH>
|
||||
<PATH>
|
||||
<PATH_ELEMENT>
|
||||
<option name="myItemId" value="crawler" />
|
||||
<option name="myItemType" value="com.intellij.ide.projectView.impl.nodes.ProjectViewProjectNode" />
|
||||
</PATH_ELEMENT>
|
||||
<PATH_ELEMENT>
|
||||
<option name="myItemId" value="crawler" />
|
||||
<option name="myItemType" value="com.intellij.ide.projectView.impl.nodes.PsiDirectoryNode" />
|
||||
</PATH_ELEMENT>
|
||||
</PATH>
|
||||
</subPane>
|
||||
</pane>
|
||||
<pane id="Scratches" />
|
||||
<pane id="Scope" />
|
||||
<pane id="PackagesPane" />
|
||||
</panes>
|
||||
</component>
|
||||
<component name="PropertiesComponent">
|
||||
<property name="settings.editor.selected.configurable" value="File.Encoding" />
|
||||
<property name="project.structure.last.edited" value="Project" />
|
||||
<property name="project.structure.proportion" value="0.15" />
|
||||
<property name="project.structure.side.proportion" value="0.0" />
|
||||
<property name="WebServerToolWindowFactoryState" value="false" />
|
||||
<property name="aspect.path.notification.shown" value="true" />
|
||||
<property name="FullScreen" value="true" />
|
||||
<property name="last_opened_file_path" value="$PROJECT_DIR$" />
|
||||
</component>
|
||||
<component name="RunDashboard">
|
||||
<option name="ruleStates">
|
||||
<list>
|
||||
<RuleState>
|
||||
<option name="name" value="ConfigurationTypeDashboardGroupingRule" />
|
||||
</RuleState>
|
||||
<RuleState>
|
||||
<option name="name" value="StatusDashboardGroupingRule" />
|
||||
</RuleState>
|
||||
</list>
|
||||
</option>
|
||||
</component>
|
||||
<component name="RunManager">
|
||||
<configuration default="true" type="#org.jetbrains.idea.devkit.run.PluginConfigurationType" factoryName="Plugin">
|
||||
<module name="" />
|
||||
<option name="VM_PARAMETERS" value="-Xmx512m -Xms256m -XX:MaxPermSize=250m -ea" />
|
||||
<option name="PROGRAM_PARAMETERS" />
|
||||
<predefined_log_file id="idea.log" enabled="true" />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="true" type="Applet" factoryName="Applet">
|
||||
<option name="HTML_USED" value="false" />
|
||||
<option name="WIDTH" value="400" />
|
||||
<option name="HEIGHT" value="300" />
|
||||
<option name="POLICY_FILE" value="$APPLICATION_HOME_DIR$/bin/appletviewer.policy" />
|
||||
<module />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="true" type="Application" factoryName="Application">
|
||||
<extension name="coverage" enabled="false" merge="false" sample_coverage="true" runner="idea" />
|
||||
<option name="MAIN_CLASS_NAME" />
|
||||
<option name="VM_PARAMETERS" />
|
||||
<option name="PROGRAM_PARAMETERS" />
|
||||
<option name="WORKING_DIRECTORY" value="$PROJECT_DIR$" />
|
||||
<option name="ALTERNATIVE_JRE_PATH_ENABLED" value="false" />
|
||||
<option name="ALTERNATIVE_JRE_PATH" />
|
||||
<option name="ENABLE_SWING_INSPECTOR" value="false" />
|
||||
<option name="ENV_VARIABLES" />
|
||||
<option name="PASS_PARENT_ENVS" value="true" />
|
||||
<module name="" />
|
||||
<envs />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="true" type="JUnit" factoryName="JUnit">
|
||||
<extension name="coverage" enabled="false" merge="false" sample_coverage="true" runner="idea" />
|
||||
<module name="" />
|
||||
<option name="ALTERNATIVE_JRE_PATH_ENABLED" value="false" />
|
||||
<option name="ALTERNATIVE_JRE_PATH" />
|
||||
<option name="PACKAGE_NAME" />
|
||||
<option name="MAIN_CLASS_NAME" />
|
||||
<option name="METHOD_NAME" />
|
||||
<option name="TEST_OBJECT" value="class" />
|
||||
<option name="VM_PARAMETERS" value="-ea" />
|
||||
<option name="PARAMETERS" />
|
||||
<option name="WORKING_DIRECTORY" value="$MODULE_DIR$" />
|
||||
<option name="ENV_VARIABLES" />
|
||||
<option name="PASS_PARENT_ENVS" value="true" />
|
||||
<option name="TEST_SEARCH_SCOPE">
|
||||
<value defaultName="singleModule" />
|
||||
</option>
|
||||
<envs />
|
||||
<patterns />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="true" type="Remote" factoryName="Remote">
|
||||
<option name="USE_SOCKET_TRANSPORT" value="true" />
|
||||
<option name="SERVER_MODE" value="false" />
|
||||
<option name="SHMEM_ADDRESS" value="javadebug" />
|
||||
<option name="HOST" value="localhost" />
|
||||
<option name="PORT" value="5005" />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="true" type="TestNG" factoryName="TestNG">
|
||||
<extension name="coverage" enabled="false" merge="false" sample_coverage="true" runner="idea" />
|
||||
<module name="" />
|
||||
<option name="ALTERNATIVE_JRE_PATH_ENABLED" value="false" />
|
||||
<option name="ALTERNATIVE_JRE_PATH" />
|
||||
<option name="SUITE_NAME" />
|
||||
<option name="PACKAGE_NAME" />
|
||||
<option name="MAIN_CLASS_NAME" />
|
||||
<option name="METHOD_NAME" />
|
||||
<option name="GROUP_NAME" />
|
||||
<option name="TEST_OBJECT" value="CLASS" />
|
||||
<option name="VM_PARAMETERS" value="-ea" />
|
||||
<option name="PARAMETERS" />
|
||||
<option name="WORKING_DIRECTORY" value="$MODULE_DIR$" />
|
||||
<option name="OUTPUT_DIRECTORY" />
|
||||
<option name="ANNOTATION_TYPE" />
|
||||
<option name="ENV_VARIABLES" />
|
||||
<option name="PASS_PARENT_ENVS" value="true" />
|
||||
<option name="TEST_SEARCH_SCOPE">
|
||||
<value defaultName="singleModule" />
|
||||
</option>
|
||||
<option name="USE_DEFAULT_REPORTERS" value="false" />
|
||||
<option name="PROPERTIES_FILE" />
|
||||
<envs />
|
||||
<properties />
|
||||
<listeners />
|
||||
<method />
|
||||
</configuration>
|
||||
</component>
|
||||
<component name="ShelveChangesManager" show_recycled="false">
|
||||
<option name="remove_strategy" value="false" />
|
||||
</component>
|
||||
<component name="TaskManager">
|
||||
<task active="true" id="Default" summary="Default task">
|
||||
<changelist id="32a73eed-3716-4f88-8891-2b38051ca637" name="Default" comment="" />
|
||||
<created>1503302070409</created>
|
||||
<option name="number" value="Default" />
|
||||
<option name="presentableId" value="Default" />
|
||||
<updated>1503302070409</updated>
|
||||
<workItem from="1503302072171" duration="53000" />
|
||||
</task>
|
||||
<servers />
|
||||
</component>
|
||||
<component name="TimeTrackingManager">
|
||||
<option name="totallyTimeSpent" value="53000" />
|
||||
</component>
|
||||
<component name="ToolWindowManager">
|
||||
<frame x="0" y="0" width="1280" height="800" extended-state="0" />
|
||||
<layout>
|
||||
<window_info id="Palette" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="TODO" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="6" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Nl-Palette" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Palette	" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Image Layers" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Capture Analysis" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Event Log" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="Maven Projects" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Run" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="2" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Version Control" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="false" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Properties" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Terminal" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Capture Tool" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Designer" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Project" active="true" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="0" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Database" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Structure" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Ant Build" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="UI Designer" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Theme Preview" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Debug" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.4" sideWeight="0.5" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Favorites" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="Cvs" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="4" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Hierarchy" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="2" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Message" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="0" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Commander" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.4" sideWeight="0.5" order="0" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Find" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Inspection" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.4" sideWeight="0.5" order="5" side_tool="false" content_ui="tabs" />
|
||||
</layout>
|
||||
</component>
|
||||
<component name="TypeScriptGeneratedFilesManager">
|
||||
<option name="processedProjectFiles" value="true" />
|
||||
</component>
|
||||
<component name="VcsContentAnnotationSettings">
|
||||
<option name="myLimit" value="2678400000" />
|
||||
</component>
|
||||
<component name="XDebuggerManager">
|
||||
<breakpoint-manager />
|
||||
<watches-manager />
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="CompilerConfiguration">
|
||||
<bytecodeTargetLevel target="1.8" />
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,71 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="MavenImportPreferences">
|
||||
<option name="generalSettings">
|
||||
<MavenGeneralSettings>
|
||||
<option name="mavenHome" value="/usr/local/Cellar/maven/3.5.0/libexec" />
|
||||
</MavenGeneralSettings>
|
||||
</option>
|
||||
</component>
|
||||
<component name="masterDetails">
|
||||
<states>
|
||||
<state key="GlobalLibrariesConfigurable.UI">
|
||||
<settings>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
<state key="JdkListConfigurable.UI">
|
||||
<settings>
|
||||
<last-edited>1.8</last-edited>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
<state key="ProjectJDKs.UI">
|
||||
<settings>
|
||||
<last-edited>1.8</last-edited>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
<state key="ProjectLibrariesConfigurable.UI">
|
||||
<settings>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
<state key="ScopeChooserConfigurable.UI">
|
||||
<settings>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
</states>
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,8 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="ProjectModuleManager">
|
||||
<modules>
|
||||
<module fileurl="file://$PROJECT_DIR$/.idea/moreSmarterCrawler.iml" filepath="$PROJECT_DIR$/.idea/moreSmarterCrawler.iml" />
|
||||
</modules>
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,9 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<module type="JAVA_MODULE" version="4">
|
||||
<component name="NewModuleRootManager" inherit-compiler-output="true">
|
||||
<exclude-output />
|
||||
<content url="file://$MODULE_DIR$" />
|
||||
<orderEntry type="inheritedJdk" />
|
||||
<orderEntry type="sourceFolder" forTests="false" />
|
||||
</component>
|
||||
</module>
|
||||
|
|
@ -0,0 +1,233 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="ChangeListManager">
|
||||
<list default="true" id="6e7ff43e-29df-49e6-9ced-009a320565f1" name="Default" comment="" />
|
||||
<option name="EXCLUDED_CONVERTED_TO_IGNORED" value="true" />
|
||||
<option name="TRACKING_ENABLED" value="true" />
|
||||
<option name="SHOW_DIALOG" value="false" />
|
||||
<option name="HIGHLIGHT_CONFLICTS" value="true" />
|
||||
<option name="HIGHLIGHT_NON_ACTIVE_CHANGELIST" value="false" />
|
||||
<option name="LAST_RESOLUTION" value="IGNORE" />
|
||||
</component>
|
||||
<component name="ExecutionTargetManager" SELECTED_TARGET="default_target" />
|
||||
<component name="GradleLocalSettings">
|
||||
<option name="externalProjectsViewState">
|
||||
<projects_view />
|
||||
</option>
|
||||
</component>
|
||||
<component name="JsBuildToolGruntFileManager" detection-done="true" sorting="DEFINITION_ORDER" />
|
||||
<component name="JsBuildToolPackageJson" detection-done="true" sorting="DEFINITION_ORDER" />
|
||||
<component name="JsGulpfileManager">
|
||||
<detection-done>true</detection-done>
|
||||
<sorting>DEFINITION_ORDER</sorting>
|
||||
</component>
|
||||
<component name="ProjectFrameBounds">
|
||||
<option name="width" value="1280" />
|
||||
<option name="height" value="800" />
|
||||
</component>
|
||||
<component name="ProjectView">
|
||||
<navigator currentView="ProjectPane" proportions="" version="1">
|
||||
<flattenPackages />
|
||||
<showMembers />
|
||||
<showModules />
|
||||
<showLibraryContents />
|
||||
<hideEmptyPackages />
|
||||
<abbreviatePackageNames />
|
||||
<autoscrollToSource />
|
||||
<autoscrollFromSource />
|
||||
<sortByType />
|
||||
<manualOrder />
|
||||
<foldersAlwaysOnTop value="true" />
|
||||
</navigator>
|
||||
<panes>
|
||||
<pane id="PackagesPane" />
|
||||
<pane id="Scratches" />
|
||||
<pane id="ProjectPane">
|
||||
<subPane>
|
||||
<PATH>
|
||||
<PATH_ELEMENT>
|
||||
<option name="myItemId" value="moreSmarterCrawler" />
|
||||
<option name="myItemType" value="com.intellij.ide.projectView.impl.nodes.ProjectViewProjectNode" />
|
||||
</PATH_ELEMENT>
|
||||
<PATH_ELEMENT>
|
||||
<option name="myItemId" value="moreSmarterCrawler" />
|
||||
<option name="myItemType" value="com.intellij.ide.projectView.impl.nodes.PsiDirectoryNode" />
|
||||
</PATH_ELEMENT>
|
||||
</PATH>
|
||||
</subPane>
|
||||
</pane>
|
||||
<pane id="Scope" />
|
||||
</panes>
|
||||
</component>
|
||||
<component name="PropertiesComponent">
|
||||
<property name="settings.editor.selected.configurable" value="File.Encoding" />
|
||||
<property name="project.structure.last.edited" value="Project" />
|
||||
<property name="project.structure.proportion" value="0.15" />
|
||||
<property name="project.structure.side.proportion" value="0.0" />
|
||||
<property name="WebServerToolWindowFactoryState" value="false" />
|
||||
<property name="aspect.path.notification.shown" value="true" />
|
||||
<property name="last_opened_file_path" value="$PROJECT_DIR$" />
|
||||
<property name="FullScreen" value="true" />
|
||||
</component>
|
||||
<component name="RunDashboard">
|
||||
<option name="ruleStates">
|
||||
<list>
|
||||
<RuleState>
|
||||
<option name="name" value="ConfigurationTypeDashboardGroupingRule" />
|
||||
</RuleState>
|
||||
<RuleState>
|
||||
<option name="name" value="StatusDashboardGroupingRule" />
|
||||
</RuleState>
|
||||
</list>
|
||||
</option>
|
||||
</component>
|
||||
<component name="RunManager">
|
||||
<configuration default="true" type="#org.jetbrains.idea.devkit.run.PluginConfigurationType" factoryName="Plugin">
|
||||
<module name="" />
|
||||
<option name="VM_PARAMETERS" value="-Xmx512m -Xms256m -XX:MaxPermSize=250m -ea" />
|
||||
<option name="PROGRAM_PARAMETERS" />
|
||||
<predefined_log_file id="idea.log" enabled="true" />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="true" type="Applet" factoryName="Applet">
|
||||
<option name="HTML_USED" value="false" />
|
||||
<option name="WIDTH" value="400" />
|
||||
<option name="HEIGHT" value="300" />
|
||||
<option name="POLICY_FILE" value="$APPLICATION_HOME_DIR$/bin/appletviewer.policy" />
|
||||
<module />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="true" type="Application" factoryName="Application">
|
||||
<extension name="coverage" enabled="false" merge="false" sample_coverage="true" runner="idea" />
|
||||
<option name="MAIN_CLASS_NAME" />
|
||||
<option name="VM_PARAMETERS" />
|
||||
<option name="PROGRAM_PARAMETERS" />
|
||||
<option name="WORKING_DIRECTORY" value="$PROJECT_DIR$" />
|
||||
<option name="ALTERNATIVE_JRE_PATH_ENABLED" value="false" />
|
||||
<option name="ALTERNATIVE_JRE_PATH" />
|
||||
<option name="ENABLE_SWING_INSPECTOR" value="false" />
|
||||
<option name="ENV_VARIABLES" />
|
||||
<option name="PASS_PARENT_ENVS" value="true" />
|
||||
<module name="" />
|
||||
<envs />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="true" type="JUnit" factoryName="JUnit">
|
||||
<extension name="coverage" enabled="false" merge="false" sample_coverage="true" runner="idea" />
|
||||
<module name="" />
|
||||
<option name="ALTERNATIVE_JRE_PATH_ENABLED" value="false" />
|
||||
<option name="ALTERNATIVE_JRE_PATH" />
|
||||
<option name="PACKAGE_NAME" />
|
||||
<option name="MAIN_CLASS_NAME" />
|
||||
<option name="METHOD_NAME" />
|
||||
<option name="TEST_OBJECT" value="class" />
|
||||
<option name="VM_PARAMETERS" value="-ea" />
|
||||
<option name="PARAMETERS" />
|
||||
<option name="WORKING_DIRECTORY" value="$MODULE_DIR$" />
|
||||
<option name="ENV_VARIABLES" />
|
||||
<option name="PASS_PARENT_ENVS" value="true" />
|
||||
<option name="TEST_SEARCH_SCOPE">
|
||||
<value defaultName="singleModule" />
|
||||
</option>
|
||||
<envs />
|
||||
<patterns />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="true" type="Remote" factoryName="Remote">
|
||||
<option name="USE_SOCKET_TRANSPORT" value="true" />
|
||||
<option name="SERVER_MODE" value="false" />
|
||||
<option name="SHMEM_ADDRESS" value="javadebug" />
|
||||
<option name="HOST" value="localhost" />
|
||||
<option name="PORT" value="5005" />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="true" type="TestNG" factoryName="TestNG">
|
||||
<extension name="coverage" enabled="false" merge="false" sample_coverage="true" runner="idea" />
|
||||
<module name="" />
|
||||
<option name="ALTERNATIVE_JRE_PATH_ENABLED" value="false" />
|
||||
<option name="ALTERNATIVE_JRE_PATH" />
|
||||
<option name="SUITE_NAME" />
|
||||
<option name="PACKAGE_NAME" />
|
||||
<option name="MAIN_CLASS_NAME" />
|
||||
<option name="METHOD_NAME" />
|
||||
<option name="GROUP_NAME" />
|
||||
<option name="TEST_OBJECT" value="CLASS" />
|
||||
<option name="VM_PARAMETERS" value="-ea" />
|
||||
<option name="PARAMETERS" />
|
||||
<option name="WORKING_DIRECTORY" value="$MODULE_DIR$" />
|
||||
<option name="OUTPUT_DIRECTORY" />
|
||||
<option name="ANNOTATION_TYPE" />
|
||||
<option name="ENV_VARIABLES" />
|
||||
<option name="PASS_PARENT_ENVS" value="true" />
|
||||
<option name="TEST_SEARCH_SCOPE">
|
||||
<value defaultName="singleModule" />
|
||||
</option>
|
||||
<option name="USE_DEFAULT_REPORTERS" value="false" />
|
||||
<option name="PROPERTIES_FILE" />
|
||||
<envs />
|
||||
<properties />
|
||||
<listeners />
|
||||
<method />
|
||||
</configuration>
|
||||
</component>
|
||||
<component name="ShelveChangesManager" show_recycled="false">
|
||||
<option name="remove_strategy" value="false" />
|
||||
</component>
|
||||
<component name="TaskManager">
|
||||
<task active="true" id="Default" summary="Default task">
|
||||
<changelist id="6e7ff43e-29df-49e6-9ced-009a320565f1" name="Default" comment="" />
|
||||
<created>1503302159478</created>
|
||||
<option name="number" value="Default" />
|
||||
<option name="presentableId" value="Default" />
|
||||
<updated>1503302159478</updated>
|
||||
<workItem from="1503302161023" duration="12000" />
|
||||
</task>
|
||||
<servers />
|
||||
</component>
|
||||
<component name="TimeTrackingManager">
|
||||
<option name="totallyTimeSpent" value="12000" />
|
||||
</component>
|
||||
<component name="ToolWindowManager">
|
||||
<frame x="0" y="0" width="1280" height="800" extended-state="0" />
|
||||
<layout>
|
||||
<window_info id="Palette" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="TODO" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="6" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Nl-Palette" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Palette	" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Image Layers" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Capture Analysis" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Event Log" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="Maven Projects" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Run" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="2" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Version Control" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="false" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Properties" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Terminal" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Capture Tool" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Designer" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Project" active="true" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="0" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Database" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Structure" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Ant Build" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="UI Designer" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Theme Preview" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Debug" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.4" sideWeight="0.5" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Favorites" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="Cvs" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="4" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Hierarchy" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="2" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Message" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="0" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Commander" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.4" sideWeight="0.5" order="0" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Find" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Inspection" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.4" sideWeight="0.5" order="5" side_tool="false" content_ui="tabs" />
|
||||
</layout>
|
||||
</component>
|
||||
<component name="TypeScriptGeneratedFilesManager">
|
||||
<option name="processedProjectFiles" value="true" />
|
||||
</component>
|
||||
<component name="VcsContentAnnotationSettings">
|
||||
<option name="myLimit" value="2678400000" />
|
||||
</component>
|
||||
<component name="XDebuggerManager">
|
||||
<breakpoint-manager />
|
||||
<watches-manager />
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,15 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="dataSourceStorageLocal">
|
||||
<data-source name="crawler@localhost" uuid="e2f306f5-80be-46ee-9e83-ac1795b7fdb7">
|
||||
<database-info product="MySQL" version="5.7.18" jdbc-version="4.0" driver-name="MySQL Connector Java" driver-version="mysql-connector-java-5.1.40 ( Revision: 402933ef52cad9aa82624e80acbea46e3a701ce6 )">
|
||||
<extra-name-characters>#@</extra-name-characters>
|
||||
<identifier-quote-string>`</identifier-quote-string>
|
||||
</database-info>
|
||||
<case-sensitivity plain-identifiers="mixed" quoted-identifiers="upper" />
|
||||
<secret-storage>master_key</secret-storage>
|
||||
<user-name>root</user-name>
|
||||
<introspection-schemas>*:crawler</introspection-schemas>
|
||||
</data-source>
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,19 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="DataSourceManagerImpl" format="xml" multifile-model="true">
|
||||
<data-source source="LOCAL" name="crawler@localhost" uuid="e2f306f5-80be-46ee-9e83-ac1795b7fdb7">
|
||||
<driver-ref>mysql</driver-ref>
|
||||
<synchronize>true</synchronize>
|
||||
<jdbc-driver>com.mysql.jdbc.Driver</jdbc-driver>
|
||||
<jdbc-url>jdbc:mysql://localhost:3306/crawler</jdbc-url>
|
||||
<driver-properties>
|
||||
<property name="autoReconnect" value="true" />
|
||||
<property name="zeroDateTimeBehavior" value="convertToNull" />
|
||||
<property name="tinyInt1isBit" value="false" />
|
||||
<property name="characterEncoding" value="utf8" />
|
||||
<property name="characterSetResults" value="utf8" />
|
||||
<property name="yearIsDateType" value="false" />
|
||||
</driver-properties>
|
||||
</data-source>
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,278 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<dataSource name="crawler@localhost">
|
||||
<database-model serializer="dbm" rdbms="MYSQL" format-version="4.2">
|
||||
<root id="1"/>
|
||||
<schema id="2" parent="1" name="crawler">
|
||||
<Current>1</Current>
|
||||
<Visible>1</Visible>
|
||||
</schema>
|
||||
<schema id="3" parent="1" name="babasport"/>
|
||||
<schema id="4" parent="1" name="customer"/>
|
||||
<schema id="5" parent="1" name="forum"/>
|
||||
<schema id="6" parent="1" name="giit"/>
|
||||
<schema id="7" parent="1" name="information_schema"/>
|
||||
<schema id="8" parent="1" name="jfinalshop"/>
|
||||
<schema id="9" parent="1" name="mysql"/>
|
||||
<schema id="10" parent="1" name="ossean"/>
|
||||
<schema id="11" parent="1" name="performance_schema"/>
|
||||
<schema id="12" parent="1" name="quick4j"/>
|
||||
<schema id="13" parent="1" name="ssm"/>
|
||||
<schema id="14" parent="1" name="ssm3"/>
|
||||
<schema id="15" parent="1" name="sys"/>
|
||||
<schema id="16" parent="1" name="test"/>
|
||||
<schema id="17" parent="1" name="extract_result"/>
|
||||
<schema id="18" parent="1" name="paper"/>
|
||||
<table id="19" parent="2" name="51cto_blog_error_page"/>
|
||||
<table id="20" parent="2" name="51cto_blog_html_detail"/>
|
||||
<table id="21" parent="2" name="51cto_blog_url"/>
|
||||
<table id="22" parent="2" name="iteye_blog_error_page"/>
|
||||
<table id="23" parent="2" name="pointers"/>
|
||||
<table id="24" parent="2" name="stackoverflow_error_page"/>
|
||||
<table id="25" parent="2" name="stackoverflow_html_detail"/>
|
||||
<table id="26" parent="2" name="stackoverflow_url"/>
|
||||
<column id="27" parent="19" name="id">
|
||||
<Position>1</Position>
|
||||
<DataType>int(11)|0</DataType>
|
||||
<NotNull>1</NotNull>
|
||||
<SequenceIdentity>1</SequenceIdentity>
|
||||
</column>
|
||||
<column id="28" parent="19" name="url">
|
||||
<Position>2</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="29" parent="19" name="html">
|
||||
<Position>3</Position>
|
||||
<DataType>mediumtext|0</DataType>
|
||||
</column>
|
||||
<column id="30" parent="19" name="crawledTime">
|
||||
<Position>4</Position>
|
||||
<DataType>datetime|0</DataType>
|
||||
</column>
|
||||
<column id="31" parent="19" name="pageMd5">
|
||||
<Position>5</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="32" parent="19" name="urlMd5">
|
||||
<Position>6</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="33" parent="19" name="type">
|
||||
<Position>7</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<key id="34" parent="19" name="PRIMARY">
|
||||
<NameSurrogate>1</NameSurrogate>
|
||||
<ColNames>id</ColNames>
|
||||
<Primary>1</Primary>
|
||||
</key>
|
||||
<column id="35" parent="20" name="id">
|
||||
<Position>1</Position>
|
||||
<DataType>int(11)|0</DataType>
|
||||
<NotNull>1</NotNull>
|
||||
<SequenceIdentity>1</SequenceIdentity>
|
||||
</column>
|
||||
<column id="36" parent="20" name="url">
|
||||
<Position>2</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="37" parent="20" name="html">
|
||||
<Position>3</Position>
|
||||
<DataType>mediumtext|0</DataType>
|
||||
</column>
|
||||
<column id="38" parent="20" name="crawledTime">
|
||||
<Position>4</Position>
|
||||
<DataType>datetime|0</DataType>
|
||||
</column>
|
||||
<column id="39" parent="20" name="urlMd5">
|
||||
<Position>5</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="40" parent="20" name="pageMd5">
|
||||
<Position>6</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="41" parent="20" name="history">
|
||||
<Position>7</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<key id="42" parent="20" name="PRIMARY">
|
||||
<NameSurrogate>1</NameSurrogate>
|
||||
<ColNames>id</ColNames>
|
||||
<Primary>1</Primary>
|
||||
</key>
|
||||
<column id="43" parent="21" name="id">
|
||||
<Position>1</Position>
|
||||
<DataType>int(11)|0</DataType>
|
||||
<NotNull>1</NotNull>
|
||||
<SequenceIdentity>1</SequenceIdentity>
|
||||
</column>
|
||||
<column id="44" parent="21" name="url">
|
||||
<Position>2</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="45" parent="21" name="timestamp">
|
||||
<Position>3</Position>
|
||||
<DataType>bigint(20)|0</DataType>
|
||||
</column>
|
||||
<column id="46" parent="21" name="extractedTime">
|
||||
<Position>4</Position>
|
||||
<DataType>datetime|0</DataType>
|
||||
</column>
|
||||
<key id="47" parent="21" name="PRIMARY">
|
||||
<NameSurrogate>1</NameSurrogate>
|
||||
<ColNames>id</ColNames>
|
||||
<Primary>1</Primary>
|
||||
</key>
|
||||
<column id="48" parent="22" name="id">
|
||||
<Position>1</Position>
|
||||
<DataType>int(11)|0</DataType>
|
||||
<NotNull>1</NotNull>
|
||||
<SequenceIdentity>1</SequenceIdentity>
|
||||
</column>
|
||||
<column id="49" parent="22" name="url">
|
||||
<Position>2</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="50" parent="22" name="html">
|
||||
<Position>3</Position>
|
||||
<DataType>mediumtext|0</DataType>
|
||||
</column>
|
||||
<column id="51" parent="22" name="crawledTime">
|
||||
<Position>4</Position>
|
||||
<DataType>datetime|0</DataType>
|
||||
</column>
|
||||
<column id="52" parent="22" name="pageMd5">
|
||||
<Position>5</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="53" parent="22" name="urlMd5">
|
||||
<Position>6</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="54" parent="22" name="type">
|
||||
<Position>7</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<key id="55" parent="22" name="PRIMARY">
|
||||
<NameSurrogate>1</NameSurrogate>
|
||||
<ColNames>id</ColNames>
|
||||
<Primary>1</Primary>
|
||||
</key>
|
||||
<column id="56" parent="23" name="id">
|
||||
<Position>1</Position>
|
||||
<DataType>int(11)|0</DataType>
|
||||
<NotNull>1</NotNull>
|
||||
<SequenceIdentity>1</SequenceIdentity>
|
||||
</column>
|
||||
<column id="57" parent="23" name="table_name">
|
||||
<Position>2</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="58" parent="23" name="pointer">
|
||||
<Position>3</Position>
|
||||
<DataType>int(11)|0</DataType>
|
||||
</column>
|
||||
<column id="59" parent="23" name="created_at">
|
||||
<Position>4</Position>
|
||||
<DataType>datetime|0</DataType>
|
||||
</column>
|
||||
<key id="60" parent="23" name="PRIMARY">
|
||||
<NameSurrogate>1</NameSurrogate>
|
||||
<ColNames>id</ColNames>
|
||||
<Primary>1</Primary>
|
||||
</key>
|
||||
<column id="61" parent="24" name="id">
|
||||
<Position>1</Position>
|
||||
<DataType>int(11)|0</DataType>
|
||||
<NotNull>1</NotNull>
|
||||
<SequenceIdentity>1</SequenceIdentity>
|
||||
</column>
|
||||
<column id="62" parent="24" name="url">
|
||||
<Position>2</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="63" parent="24" name="html">
|
||||
<Position>3</Position>
|
||||
<DataType>mediumtext|0</DataType>
|
||||
</column>
|
||||
<column id="64" parent="24" name="crawledTime">
|
||||
<Position>4</Position>
|
||||
<DataType>datetime|0</DataType>
|
||||
</column>
|
||||
<column id="65" parent="24" name="pageMd5">
|
||||
<Position>5</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="66" parent="24" name="urlMd5">
|
||||
<Position>6</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="67" parent="24" name="type">
|
||||
<Position>7</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<key id="68" parent="24" name="PRIMARY">
|
||||
<NameSurrogate>1</NameSurrogate>
|
||||
<ColNames>id</ColNames>
|
||||
<Primary>1</Primary>
|
||||
</key>
|
||||
<column id="69" parent="25" name="id">
|
||||
<Position>1</Position>
|
||||
<DataType>int(11)|0</DataType>
|
||||
<NotNull>1</NotNull>
|
||||
<SequenceIdentity>1</SequenceIdentity>
|
||||
</column>
|
||||
<column id="70" parent="25" name="url">
|
||||
<Position>2</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="71" parent="25" name="html">
|
||||
<Position>3</Position>
|
||||
<DataType>mediumtext|0</DataType>
|
||||
</column>
|
||||
<column id="72" parent="25" name="crawledTime">
|
||||
<Position>4</Position>
|
||||
<DataType>datetime|0</DataType>
|
||||
</column>
|
||||
<column id="73" parent="25" name="urlMd5">
|
||||
<Position>5</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="74" parent="25" name="pageMd5">
|
||||
<Position>6</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="75" parent="25" name="history">
|
||||
<Position>7</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<key id="76" parent="25" name="PRIMARY">
|
||||
<NameSurrogate>1</NameSurrogate>
|
||||
<ColNames>id</ColNames>
|
||||
<Primary>1</Primary>
|
||||
</key>
|
||||
<column id="77" parent="26" name="id">
|
||||
<Position>1</Position>
|
||||
<DataType>int(11)|0</DataType>
|
||||
<NotNull>1</NotNull>
|
||||
<SequenceIdentity>1</SequenceIdentity>
|
||||
</column>
|
||||
<column id="78" parent="26" name="url">
|
||||
<Position>2</Position>
|
||||
<DataType>varchar(255)|0</DataType>
|
||||
</column>
|
||||
<column id="79" parent="26" name="timestamp">
|
||||
<Position>3</Position>
|
||||
<DataType>bigint(20)|0</DataType>
|
||||
</column>
|
||||
<column id="80" parent="26" name="extractedTime">
|
||||
<Position>4</Position>
|
||||
<DataType>datetime|0</DataType>
|
||||
</column>
|
||||
<key id="81" parent="26" name="PRIMARY">
|
||||
<NameSurrogate>1</NameSurrogate>
|
||||
<ColNames>id</ColNames>
|
||||
<Primary>1</Primary>
|
||||
</key>
|
||||
</database-model>
|
||||
</dataSource>
|
||||
|
|
@ -10,20 +10,4 @@
|
|||
<component name="ProjectRootManager" version="2" languageLevel="JDK_1_8" default="true" project-jdk-name="1.8" project-jdk-type="JavaSDK">
|
||||
<output url="file://$PROJECT_DIR$/classes" />
|
||||
</component>
|
||||
<component name="masterDetails">
|
||||
<states>
|
||||
<state key="ProjectJDKs.UI">
|
||||
<settings>
|
||||
<last-edited>1.8</last-edited>
|
||||
<splitter-proportions>
|
||||
<option name="proportions">
|
||||
<list>
|
||||
<option value="0.2" />
|
||||
</list>
|
||||
</option>
|
||||
</splitter-proportions>
|
||||
</settings>
|
||||
</state>
|
||||
</states>
|
||||
</component>
|
||||
</project>
|
||||
File diff suppressed because it is too large
Load Diff
|
|
@ -1,6 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<module org.jetbrains.idea.maven.project.MavenProjectsManager.isMavenModule="true" type="JAVA_MODULE" version="4">
|
||||
<component name="NewModuleRootManager" LANGUAGE_LEVEL="JDK_1_7" inherit-compiler-output="false">
|
||||
<component name="NewModuleRootManager" LANGUAGE_LEVEL="JDK_1_7">
|
||||
<output url="file://$MODULE_DIR$/target/classes" />
|
||||
<output-test url="file://$MODULE_DIR$/target/test-classes" />
|
||||
<content url="file://$MODULE_DIR$">
|
||||
|
|
|
|||
|
|
@ -16,11 +16,11 @@
|
|||
destroy-method="close">
|
||||
<property name="driverClassName" value="com.mysql.jdbc.Driver" />
|
||||
<property name="url"
|
||||
value="jdbc:mysql://localhost:3306/zzx_crawler?characterEncoding=UTF-8" />
|
||||
value="jdbc:mysql://localhost:3306/crawler?characterEncoding=UTF-8" />
|
||||
<property name="username" value="root" />
|
||||
<!--<property name="password" value="1234" /> -->
|
||||
|
||||
<property name="password" value="123456" />
|
||||
<property name="password" value="root" />
|
||||
</bean>
|
||||
|
||||
</beans>
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: cnblogs_news
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'https://news.cnblogs.com/n/page/{}/'
|
||||
|
||||
self.guess_num = 100
|
||||
self.guess_span = 10
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i < 10:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=1*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.news_entry > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaBlog(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'cnblogs_news'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,133 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: cnblogs_news
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'https://news.cnblogs.com/n/page/{}/'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=1*60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'https://news.cnblogs.com/'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'https://news.cnblogs.com/n/page/1/'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
full_num = int(response.doc('#pager > a:nth-last-child(2)').text())
|
||||
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.news_entry > a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = CnblogsNews(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class CnblogsNews(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'cnblogs_news_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: cnblogs_q_solved
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'https://q.cnblogs.com/list/solved?page={}'
|
||||
|
||||
self.guess_num = 161
|
||||
self.guess_span = 10
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=1*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.news_item > h2 > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaBlog(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'cnblogs_q_solved'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,133 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: cnblogs_q_solved
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'https://q.cnblogs.com/list/solved?page={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'https://q.cnblogs.com/list/solved'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'https://q.cnblogs.com/list/solved?page=1'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
full_num = int(response.doc('#pager > a:nth-last-child(2)').text())
|
||||
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.news_item > h2 > a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = CnblogsQSolved(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class CnblogsQSolved(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'cnblogs_q_solved_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: cnblogs_q_unsolved
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'https://q.cnblogs.com/list/unsolved?page={}'
|
||||
|
||||
self.guess_num = 161
|
||||
self.guess_span = 10
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=1*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.news_item > h2 > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaBlog(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'cnblogs_q_unsolved'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,133 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: cnblogs_q_solved
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'https://q.cnblogs.com/list/unsolved?page={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'https://q.cnblogs.com/list/unsolved'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'https://q.cnblogs.com/list/unsolved?page=1'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
full_num = int(response.doc('#pager > a:nth-last-child(2)').text())
|
||||
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.news_item > h2 > a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = CnblogsQUnsolved(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class CnblogsQUnsolved(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'cnblogs_q_unsolved_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: csdn_ask
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'http://ask.csdn.net/p{}'
|
||||
|
||||
self.guess_num = 1000
|
||||
self.guess_span = 500
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i < 10:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=1*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.questions_detail_con dt a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaBlog(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'csdn_ask'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,135 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: csdn_ask
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'http://ask.csdn.net/p{}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5*60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'http://ask.csdn.net/'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'http://ask.csdn.net/'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
num_span = response.doc('.page-nav a:last-child').attr.href
|
||||
pattern = re.compile(r'http://ask.csdn.net/p(.*)')
|
||||
full_num = int(pattern.findall(num_span)[0])
|
||||
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.questions_detail_con>dl>dt>a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = CsdnAsk(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class CsdnAsk(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'csdn_ask_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: csdn_blog
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'http://blog.csdn.net/?&page={}'
|
||||
|
||||
self.guess_num = 340
|
||||
self.guess_span = 10
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=1*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.blog_list .tracking-ad > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaBlog(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'csdn_blog'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,135 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: csdn_blog
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'http://blog.csdn.net/other/newarticle.html?&page={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'http://blog.csdn.net/other/newarticle.html'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'http://blog.csdn.net/other/newarticle.html?&page=1'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
num_span = response.doc('body div.page_nav > span').text()
|
||||
pattern = re.compile(r'共(.*)页')
|
||||
full_num = int(pattern.findall(num_span)[0])
|
||||
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.blog_list .tracking-ad > a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = CsdnBlog(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class CsdnBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'csdn_blog_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,140 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: csdn_topic
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.sites = ['Mobile','CloudComputing','Java','DotNET','WebDevelop','Linux']
|
||||
self.base_url = 'http://bbs.csdn.net/forums/{}?page={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
for site in self.sites:
|
||||
list_url = 'http://bbs.csdn.net/forums/{}'.format(site)
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
first_page = 'http://bbs.csdn.net/forums/{}?page=1'
|
||||
for site in self.sites:
|
||||
self.crawl(first_page.format(site), callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
|
||||
def get_full(self, response):
|
||||
num_text = next(response.doc('.page_nav > ul > li:nth-last-child(1) > span:nth-child(2)').items()).text()
|
||||
pattern = re.compile(r'共(.*)页')
|
||||
full_num = int(pattern.findall(num_text)[0])
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('div.content tr')
|
||||
for each in targets.items():
|
||||
target = each('.title > a').attr.href
|
||||
flag = each('td:nth-last-child(2) > span.time').text()
|
||||
if target is not None:
|
||||
self.crawl(target, callback=self.detail_page, headers=self.headers, itag=flag)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = CsdnTopic(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class CsdnTopic(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'csdn_topic_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,65 @@
|
|||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.sql import func
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
|
||||
class Table(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'stackoverflow_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
class Pointers(Base):
|
||||
__tablename__ = 'pointers'
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
table_name = Column(String(255))
|
||||
pointer = Column(INTEGER)
|
||||
created_at = Column(DATETIME)
|
||||
|
||||
|
||||
class Url(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'stackoverflow_url'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
timestamp = Column(BIGINT)
|
||||
extractedTime = Column(DATETIME)
|
||||
|
||||
|
||||
engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/crawler')
|
||||
DBSession = sessionmaker(bind=engine)
|
||||
Base.metadata.create_all(engine)
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
# session = DBSession()
|
||||
# pointer = session.query(Pointers.pointer).filter(Pointers.table_name=='stackoverflow_url').one()[0]
|
||||
# url_num = session.query(func.max(Url.id)).one()[0]
|
||||
# url = session.query(Url.url).filter(Url.id==pointer+1).one()[0]
|
||||
# session.query(Pointers).filter(Pointers.table_name=='stackoverflow_url').update({Pointers.pointer:pointer+1})
|
||||
# session.commit()
|
||||
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: iteye_ask
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'http://www.iteye.com/problems/hot?page={}'
|
||||
|
||||
self.guess_num = 1000
|
||||
self.guess_span = 500
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=1*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.question-summary .summary > h3 > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = Table(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class Table(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'iteye_ask'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,133 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: iteye_ask
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'http://www.iteye.com/ask?page={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=24*60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'http://www.iteye.com/ask'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'http://www.iteye.com/ask?page=1'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
full_num = int(response.doc('.pagination > a:nth-last-child(2)').text())
|
||||
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.question-summary .summary > h3 > a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = IteyeAsk(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class IteyeAsk(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'iteye_ask_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,133 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: iteye_blog
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'http://www.iteye.com/blogs?page={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=24*60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'http://www.iteye.com/blogs'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'http://www.iteye.com/blogs?page=1'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
full_num = int(response.doc('div.pagination > a:nth-last-child(2)').text())
|
||||
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.content h3 a[title]')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = IteyeBlog(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class IteyeBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'iteye_blog_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: oschina_blog
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'http://www.iteye.com/news?page={}'
|
||||
|
||||
self.guess_num = 750
|
||||
self.guess_span = 10
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=5*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.content h3 a[title]')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaBlog(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'iteye_news'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,133 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: iteye_news
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'http://www.iteye.com/news?page={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=24*60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'http://www.iteye.com/news'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'http://www.iteye.com/news?page=1'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
full_num = int(response.doc('div.pagination > a:nth-last-child(2)').text())
|
||||
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.content h3 a[title]')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = IteyeNews(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class IteyeNews(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'iteye_news_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: openhub
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'https://www.openhub.net/p?page={}&query=&ref=explore_project'
|
||||
|
||||
self.guess_num = 1000
|
||||
self.guess_span = 1000
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=24 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=12*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.well.searchable .title.pull-left > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = Table(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class Table(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'openhub'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,135 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: openhub
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'https://www.openhub.net/p?page={}&query=&sort=new'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=3*60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'https://www.openhub.net/p?page=1&query=&sort=new'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'https://www.openhub.net/p?page=1&query=&sort=new'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
full_num = int(response.doc('ul.pagination > li:nth-last-child(2) > a').text())
|
||||
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.well.searchable')
|
||||
for each in targets.items():
|
||||
target = each('.title.pull-left > a').attr.href
|
||||
flag = each('#inner_content > div.stats.pull-left > p:nth-child(1) >a').text() # 内容增量
|
||||
self.crawl(target, callback=self.detail_page, headers=self.headers, itag=flag)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = Openhub(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class Openhub(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'openhub_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,148 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: oschina_blog
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'https://www.oschina.net/action/ajax/get_more_recommend_blog?classification=0&p={}'
|
||||
|
||||
#希望找到一个比较通用的,对于ajax动态加载页面遍历页数的方法。设定一个很大的页数是一个办法,但是不够灵活。
|
||||
|
||||
# 开始时,猜测列表页总页数为499页(经历第一轮的不断爬取后,这个数据会更新成更合理的数据)
|
||||
self.guess_num = 500
|
||||
# 如果前500页未包括所有的列表页,那么再往下猜100页
|
||||
self.guess_span = 100
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=24 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=24*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
#targets是列表页中的条目
|
||||
targets = response.doc('header > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
# self.not_blank<=self.looked,当self.looked+1==self.guess_num时,可以肯定,guess_num范围内的列表页都爬完了
|
||||
# self.bot_blank+1==self.guess_num,说明已经爬完了guess_num范围内的页面,并且没有空的列表页
|
||||
# 如果所有列表页都非空,则视为没有爬完全部列表页(guess_num不为总数)
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=10*24*60*60, headers=self.headers)
|
||||
# 如果not_blank+1的数量不等于guess_num的数量,则有可能是还没爬完guess_num范围,也有可能是已经爬完guess_num范围的页面了但是存在空页
|
||||
# 如果是已经爬完guess_num的范围内页面了,并且存在空页的情况,则应该让guess_num等全局属性值复原
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
# self.guess_num 不应该复原,因为下次on_start方法可以直接使用新的更符合情况的guess_num值
|
||||
# self.guess_num = 500
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaBlog(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'oschina_blog'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,133 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: oschina_blog
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'https://www.oschina.net/action/ajax/get_more_recent_blog?classification=0&p={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=30)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'https://www.oschina.net/action/ajax/get_more_recent_blog?classification=0'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
# def full(self):
|
||||
# first_page = 'https://www.oschina.net/action/ajax/get_more_recent_blog?classification=0&p=1'
|
||||
# self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
# self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
# def get_full(self, response):
|
||||
# full_num = int(response.doc('div.pagination > a:nth-last-child(2)').text())
|
||||
|
||||
# for i in range(2, full_num+1):
|
||||
# list_url = self.base_url.format(str(i))
|
||||
# self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('header > a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaBlog(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'oschina_blog_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,136 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: oschina_blog
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'https://www.oschina.net/action/ajax/get_more_news_list?newsType=&p={}'
|
||||
|
||||
self.guess_num = 500
|
||||
self.guess_span = 100
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=5*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.main-info .title')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc.text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaBlog(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaBlog(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'oschina_news'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,140 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: oschina_project
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'https://www.oschina.net/project/list?company=0&sort=score&lang=0&recommend=false&p={}'
|
||||
|
||||
self.guess_num = 1000
|
||||
self.guess_span = 500
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=24 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=12*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.box.item .box-aw > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = Table(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class Table(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'oschina_project'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,133 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: oschina_project
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
import re
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'https://www.oschina.net/project/list?sort=time&p={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=24*60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'https://www.oschina.net/project/list?sort=time'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'https://www.oschina.net/project/list?sort=time&p=1'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
full_num = int(response.doc('ul.paging > li:nth-last-child(2) > a').text())
|
||||
|
||||
for i in range(2, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.box.item .box-aw > a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaProject(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaProject(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'oschina_project_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: oschina_question
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'https://www.oschina.net/question/_widgets/_list_content?catalog=1&show=active&p={}'
|
||||
|
||||
self.guess_num = 1000
|
||||
self.guess_span = 500
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=1*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.box-aw.question_detail .title > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = Table(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class Table(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'oschina_question'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,155 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: oschina_question
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'https://www.oschina.net/question?catalog=1&show=time&p={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
#全量模式辅助变量
|
||||
self.guess_num = 500
|
||||
self.guess_span = 500
|
||||
self.not_blank = 0 #非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.looked = 0 #所有处理过的列表页计数
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'https://www.oschina.net/question?catalog=1&show=time'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
for i in range(1, self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page_full, retries=0, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page_full(self, response):
|
||||
targets = response.doc('.box-aw.question_detail .title > a')
|
||||
self.looked += 1 # self.looked:每次调用该函数就会加1,表示出现了一个response,即爬了一个列表页
|
||||
if targets.length > 0:
|
||||
self.not_blank += 1 # 处理的列表页中,不是空列表页的
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if self.looked == self.guess_num-1 and self.looked == self.not_blank: # 处理过的列表页均非空,说明很可能后面还有列表页
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page_full, retries=0, headers=self.headers)
|
||||
elif self.looked == self.guess_num-1:
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.box-aw.question_detail .title > a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = OschinaQuestion(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class OschinaQuestion(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'oschina_question_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,137 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: stackoverflow
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.sql import func
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表,有就忽略
|
||||
|
||||
def on_start(self):
|
||||
session = self.DBSession()
|
||||
pointer = session.query(Pointers.pointer).filter(Pointers.table_name=='stackoverflow_url').one()[0]
|
||||
url_num = session.query(func.max(Url.id)).one()[0]
|
||||
|
||||
for i in range(pointer+1, url_num+1):
|
||||
url = session.query(Url.url).filter(Url.id==i).one()[0]
|
||||
print(url)
|
||||
self.crawl(url, callback=self.detail_page, headers=self.headers, retries=0)
|
||||
|
||||
# 每增一百条就更新下指针
|
||||
if (i-pointer)%100==0 or url_num-i<100:
|
||||
session.query(Pointers).filter(Pointers.table_name=='stackoverflow_url').update({Pointers.pointer:i})
|
||||
session.commit()
|
||||
|
||||
session.close()
|
||||
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = Table(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
# def update_pointer(self):
|
||||
# session = self.DBSession()
|
||||
# pointer = session.query(Pointers.pointer).filter(Pointers.table_name=='stackoverflow_url').one()[0]
|
||||
# session.query(Pointers).filter(Pointers.table_name=='stackoverflow_url').update({Pointers.pointer:pointer+1})
|
||||
# session.commit()
|
||||
# session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
|
||||
Base = declarative_base()
|
||||
|
||||
|
||||
class Table(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'stackoverflow_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
class Pointers(Base):
|
||||
__tablename__ = 'pointers'
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
table_name = Column(String(255))
|
||||
pointer = Column(INTEGER)
|
||||
created_at = Column(DATETIME)
|
||||
|
||||
|
||||
class Url(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'stackoverflow_url'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
timestamp = Column(BIGINT)
|
||||
extractedTime = Column(DATETIME)
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: sourceforge
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'https://sourceforge.net/directory/?page={}'
|
||||
|
||||
self.guess_num = 1000
|
||||
self.guess_span = 1000
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=24 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i == 1:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=12*60*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.projects li > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = Table(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class Table(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'sourceforge'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,155 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: sourceforge
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 2*60,
|
||||
3: 3*60,
|
||||
4: 4*60,
|
||||
5: 10*60,
|
||||
6: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'https://sourceforge.net/directory/freshness%3Arecently-updated/?sort=update&page={}'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
#全量模式辅助变量
|
||||
self.guess_num = 500
|
||||
self.guess_span = 500
|
||||
self.not_blank = 0 #非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.looked = 0 #所有处理过的列表页计数
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://influx:influx1234@192.168.80.104:3306/pages')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=60)
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
list_url = 'https://sourceforge.net/directory/freshness%3Arecently-updated/?sort=update'
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, priority=2, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
for i in range(1, self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page_full, retries=5, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page_full(self, response):
|
||||
targets = response.doc('.projects li > a')
|
||||
|
||||
self.looked += 1 # self.looked:每次调用该函数就会加1,表示出现了一个response,即爬了一个列表页
|
||||
if targets.length > 0:
|
||||
self.not_blank += 1 # 处理的列表页中,不是空列表页的
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if self.looked == self.guess_num-1 and self.looked == self.not_blank: # 处理过的列表页均非空,说明很可能后面还有列表页
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page_full, retries=5, headers=self.headers)
|
||||
elif self.looked == self.guess_num-1:
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.projects li > a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
@config(priority=1)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = Sourceforge(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class Sourceforge(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'sourceforge_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,139 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: stackoverflow
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
self.base_url = 'https://stackoverflow.com/questions?page={}&sort=newest'
|
||||
|
||||
self.guess_num = 1000
|
||||
self.guess_span = 1000
|
||||
|
||||
# 非空列表页计数,每遇到一个非空列表页,+1
|
||||
self.not_blank = 0
|
||||
# 所有处理过的列表页计数
|
||||
self.looked = 0
|
||||
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=1 * 60)
|
||||
def on_start(self):
|
||||
for i in range(1,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
# 第一页需要更加频繁地进行爬取
|
||||
if i < 10:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, age=30*60, headers=self.headers)
|
||||
else:
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
|
||||
targets = response.doc('.summary h3 > a')
|
||||
|
||||
# 所有页计数,无论空不空,都+1
|
||||
self.looked += 1
|
||||
|
||||
# 非空页判定
|
||||
if(targets.length > 0):
|
||||
self.not_blank += 1
|
||||
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
if(self.not_blank+1 == self.guess_num):
|
||||
tmp = self.guess_num
|
||||
self.guess_num += self.guess_span
|
||||
for i in range(tmp,self.guess_num):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=10, headers=self.headers)
|
||||
|
||||
elif(self.looked+1 == self.guess_num):
|
||||
self.not_blank = 0
|
||||
self.looked = 0
|
||||
else:
|
||||
pass
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
# "title": response.doc('.blog-content .title').remove('span').text(), #把标题中的“原”,“荐”等无关信息删除掉
|
||||
# "html": response.doc('html').text(),
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMD5": self.get_md5(response.url.encode()),
|
||||
"pageMD5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = Table(url=result['url'],html=result['html'],
|
||||
urlMD5=result['urlMD5'],crawledTime=result['crawledTime'],
|
||||
pageMD5=result['pageMD5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class Table(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'stackoverflow'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMD5 = Column(String(255))
|
||||
pageMD5 = Column(String(255),)
|
||||
history = Column(String(255))
|
||||
|
||||
|
|
@ -0,0 +1,131 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
# Created on 2017-03-20 10:08:31
|
||||
# Project: stackoverflow
|
||||
|
||||
from pyspider.libs.base_handler import *
|
||||
|
||||
import hashlib
|
||||
from datetime import *
|
||||
|
||||
from sqlalchemy import Column, String, create_engine
|
||||
from sqlalchemy.orm import sessionmaker
|
||||
from sqlalchemy.ext.declarative import declarative_base
|
||||
from sqlalchemy.dialects.mysql import \
|
||||
BIGINT, BINARY, BIT, BLOB, BOOLEAN, CHAR, DATE, \
|
||||
DATETIME, DECIMAL, DECIMAL, DOUBLE, ENUM, FLOAT, INTEGER, \
|
||||
LONGBLOB, LONGTEXT, MEDIUMBLOB, MEDIUMINT, MEDIUMTEXT, NCHAR, \
|
||||
NUMERIC, NVARCHAR, REAL, SET, SMALLINT, TEXT, TIME, TIMESTAMP, \
|
||||
TINYBLOB, TINYINT, TINYTEXT, VARBINARY, VARCHAR, YEAR
|
||||
|
||||
class Handler(BaseHandler):
|
||||
crawl_config = {
|
||||
'itag':'v1'
|
||||
}
|
||||
|
||||
retry_delay = {
|
||||
0: 0,
|
||||
1: 60,
|
||||
2: 5*60,
|
||||
3: 10*60,
|
||||
4: 30*60,
|
||||
'': 60*60
|
||||
}
|
||||
|
||||
def __init__(self):
|
||||
self.headers = {'User-Agent':'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 '
|
||||
'(KHTML, like Gecko) Chrome/57.0.2987.98 Safari/537.36'}
|
||||
|
||||
self.base_url = 'https://stackoverflow.com/questions?page={}&sort=newest'
|
||||
|
||||
self.mode = 1 # 默认为增量模式
|
||||
|
||||
# 数据库相关
|
||||
self.engine = create_engine('mysql+mysqlconnector://root:root@localhost:3306/ossean')
|
||||
self.DBSession = sessionmaker(bind=self.engine)
|
||||
Base.metadata.create_all(self.engine) #自建数据库表
|
||||
|
||||
@every(minutes=5) #每5分钟增量爬一次
|
||||
def on_start(self):
|
||||
if self.mode == 1:
|
||||
self.increment()
|
||||
else:
|
||||
self.full()
|
||||
|
||||
|
||||
def increment(self):
|
||||
for i in range(1, 5):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=5, age=1*60, headers=self.headers)
|
||||
|
||||
|
||||
def full(self):
|
||||
first_page = 'https://stackoverflow.com/questions'
|
||||
self.crawl(first_page, callback=self.get_full, retries=5, age=1*60, headers=self.headers)
|
||||
self.mode = 1 #确保每次on_start重启时是处在增量模式
|
||||
|
||||
def get_full(self, response):
|
||||
full_num = int(response.doc('#mainbar > div.pager.fl > a:nth-child(7) > span').text()) #列表页总数
|
||||
|
||||
for i in range(5, full_num+1):
|
||||
list_url = self.base_url.format(str(i))
|
||||
self.crawl(list_url, callback=self.index_page, retries=0, headers=self.headers)
|
||||
|
||||
|
||||
@config(age=10 * 24 * 60 * 60)
|
||||
def index_page(self, response):
|
||||
targets = response.doc('.summary h3 > a')
|
||||
for each in targets.items():
|
||||
self.crawl(each.attr.href, callback=self.detail_page, headers=self.headers)
|
||||
|
||||
|
||||
@config(priority=2)
|
||||
def detail_page(self, response):
|
||||
return {
|
||||
"url": response.url,
|
||||
"html": response.text,
|
||||
"crawledTime": datetime.now().strftime('%Y-%m-%d %H:%M:%S'),
|
||||
"urlMd5": self.get_md5(response.url.encode()),
|
||||
"pageMd5":self.get_md5(response.text.encode()),
|
||||
}
|
||||
|
||||
|
||||
def on_result(self,result):
|
||||
if not result or not str(result['url']): #如果未返回结果或者返回结果的url字段为空,则不做处理
|
||||
return
|
||||
|
||||
session = self.DBSession() #存库所需
|
||||
|
||||
#将result格式化成sqlalchemy能够接受的数据类型(字典转列表)
|
||||
data = Table(url=result['url'],html=result['html'],
|
||||
urlMd5=result['urlMd5'],crawledTime=result['crawledTime'],
|
||||
pageMd5=result['pageMd5']
|
||||
)
|
||||
|
||||
#存库
|
||||
session.merge(data)
|
||||
session.commit()
|
||||
session.close()
|
||||
|
||||
def get_md5(self,data):
|
||||
m = hashlib.md5()
|
||||
m.update(data)
|
||||
return m.hexdigest()
|
||||
|
||||
#数据库相关操作
|
||||
Base = declarative_base()
|
||||
|
||||
class Table(Base):
|
||||
# 表的名字:
|
||||
__tablename__ = 'stackoverflow_html_detail'
|
||||
|
||||
# 表的结构:
|
||||
id = Column(INTEGER, primary_key=True)
|
||||
url = Column(String(255), nullable=False)
|
||||
html = Column(MEDIUMTEXT)
|
||||
crawledTime = Column(DATETIME)
|
||||
urlMd5 = Column(String(255))
|
||||
pageMd5 = Column(String(255))
|
||||
history = Column(String(255))
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,10 @@
|
|||
#!/usr/bin/env python
|
||||
# -*- encoding: utf-8 -*-
|
||||
|
||||
sites = ['Mobile','CloudComputing','Java','DotNET','WebDevelop','Linux']
|
||||
|
||||
base_url = 'http://bbs.csdn.net/forums/{}?page={}'
|
||||
|
||||
for site in sites:
|
||||
url = base_url.format(site,1)
|
||||
print(url)
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
|
|
@ -0,0 +1,16 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="CompilerConfiguration">
|
||||
<annotationProcessing>
|
||||
<profile name="Maven default annotation processors profile" enabled="true">
|
||||
<sourceOutputDir name="target/generated-sources/annotations" />
|
||||
<sourceTestOutputDir name="target/generated-test-sources/test-annotations" />
|
||||
<outputRelativeToContentRoot value="true" />
|
||||
<module name="gather_program" />
|
||||
</profile>
|
||||
</annotationProcessing>
|
||||
<bytecodeTargetLevel>
|
||||
<module name="gather_program" target="1.7" />
|
||||
</bytecodeTargetLevel>
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="Encoding">
|
||||
<file url="file://$PROJECT_DIR$" charset="UTF-8" />
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: aopalliance:aopalliance:1.0">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/aopalliance/aopalliance/1.0/aopalliance-1.0.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/aopalliance/aopalliance/1.0/aopalliance-1.0-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/aopalliance/aopalliance/1.0/aopalliance-1.0-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: commons-collections:commons-collections:3.2.1">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-collections/commons-collections/3.2.1/commons-collections-3.2.1.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-collections/commons-collections/3.2.1/commons-collections-3.2.1-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-collections/commons-collections/3.2.1/commons-collections-3.2.1-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: commons-dbcp:commons-dbcp:1.3">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-dbcp/commons-dbcp/1.3/commons-dbcp-1.3.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-dbcp/commons-dbcp/1.3/commons-dbcp-1.3-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-dbcp/commons-dbcp/1.3/commons-dbcp-1.3-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: commons-io:commons-io:1.3.2">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-io/commons-io/1.3.2/commons-io-1.3.2.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-io/commons-io/1.3.2/commons-io-1.3.2-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-io/commons-io/1.3.2/commons-io-1.3.2-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: commons-logging:commons-logging:1.2">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-logging/commons-logging/1.2/commons-logging-1.2.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-logging/commons-logging/1.2/commons-logging-1.2-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-logging/commons-logging/1.2/commons-logging-1.2-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: commons-pool:commons-pool:1.5.4">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-pool/commons-pool/1.5.4/commons-pool-1.5.4.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-pool/commons-pool/1.5.4/commons-pool-1.5.4-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-pool/commons-pool/1.5.4/commons-pool-1.5.4-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: junit:junit:3.8.1">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/junit/junit/3.8.1/junit-3.8.1.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/junit/junit/3.8.1/junit-3.8.1-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/junit/junit/3.8.1/junit-3.8.1-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: log4j:log4j:1.2.17">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/log4j/log4j/1.2.17/log4j-1.2.17.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/log4j/log4j/1.2.17/log4j-1.2.17-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/log4j/log4j/1.2.17/log4j-1.2.17-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: mysql:mysql-connector-java:5.1.18">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/mysql/mysql-connector-java/5.1.18/mysql-connector-java-5.1.18.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/mysql/mysql-connector-java/5.1.18/mysql-connector-java-5.1.18-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/mysql/mysql-connector-java/5.1.18/mysql-connector-java-5.1.18-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.apache.commons:commons-lang3:3.1">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/apache/commons/commons-lang3/3.1/commons-lang3-3.1.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/apache/commons/commons-lang3/3.1/commons-lang3-3.1-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/apache/commons/commons-lang3/3.1/commons-lang3-3.1-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.mybatis:mybatis:3.1.1">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/mybatis/mybatis/3.1.1/mybatis-3.1.1.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/mybatis/mybatis/3.1.1/mybatis-3.1.1-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/mybatis/mybatis/3.1.1/mybatis-3.1.1-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.mybatis:mybatis-spring:1.1.1">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/mybatis/mybatis-spring/1.1.1/mybatis-spring-1.1.1.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/mybatis/mybatis-spring/1.1.1/mybatis-spring-1.1.1-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/mybatis/mybatis-spring/1.1.1/mybatis-spring-1.1.1-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.slf4j:slf4j-api:1.7.7">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/slf4j/slf4j-api/1.7.7/slf4j-api-1.7.7.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/slf4j/slf4j-api/1.7.7/slf4j-api-1.7.7-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/slf4j/slf4j-api/1.7.7/slf4j-api-1.7.7-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.slf4j:slf4j-log4j12:1.7.7">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/slf4j/slf4j-log4j12/1.7.7/slf4j-log4j12-1.7.7.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/slf4j/slf4j-log4j12/1.7.7/slf4j-log4j12-1.7.7-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/slf4j/slf4j-log4j12/1.7.7/slf4j-log4j12-1.7.7-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.springframework:spring-aop:4.1.4.RELEASE">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-aop/4.1.4.RELEASE/spring-aop-4.1.4.RELEASE.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-aop/4.1.4.RELEASE/spring-aop-4.1.4.RELEASE-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-aop/4.1.4.RELEASE/spring-aop-4.1.4.RELEASE-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.springframework:spring-beans:4.1.4.RELEASE">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-beans/4.1.4.RELEASE/spring-beans-4.1.4.RELEASE.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-beans/4.1.4.RELEASE/spring-beans-4.1.4.RELEASE-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-beans/4.1.4.RELEASE/spring-beans-4.1.4.RELEASE-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.springframework:spring-context:4.1.4.RELEASE">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-context/4.1.4.RELEASE/spring-context-4.1.4.RELEASE.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-context/4.1.4.RELEASE/spring-context-4.1.4.RELEASE-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-context/4.1.4.RELEASE/spring-context-4.1.4.RELEASE-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.springframework:spring-core:4.1.4.RELEASE">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-core/4.1.4.RELEASE/spring-core-4.1.4.RELEASE.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-core/4.1.4.RELEASE/spring-core-4.1.4.RELEASE-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-core/4.1.4.RELEASE/spring-core-4.1.4.RELEASE-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.springframework:spring-expression:4.1.4.RELEASE">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-expression/4.1.4.RELEASE/spring-expression-4.1.4.RELEASE.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-expression/4.1.4.RELEASE/spring-expression-4.1.4.RELEASE-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-expression/4.1.4.RELEASE/spring-expression-4.1.4.RELEASE-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.springframework:spring-jdbc:3.1.1.RELEASE">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-jdbc/3.1.1.RELEASE/spring-jdbc-3.1.1.RELEASE.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-jdbc/3.1.1.RELEASE/spring-jdbc-3.1.1.RELEASE-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-jdbc/3.1.1.RELEASE/spring-jdbc-3.1.1.RELEASE-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.springframework:spring-tx:3.1.1.RELEASE">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-tx/3.1.1.RELEASE/spring-tx-3.1.1.RELEASE.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-tx/3.1.1.RELEASE/spring-tx-3.1.1.RELEASE-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/springframework/spring-tx/3.1.1.RELEASE/spring-tx-3.1.1.RELEASE-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,8 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="ProjectModuleManager">
|
||||
<modules>
|
||||
<module fileurl="file://$PROJECT_DIR$/gather_program.iml" filepath="$PROJECT_DIR$/gather_program.iml" />
|
||||
</modules>
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -9,6 +9,9 @@
|
|||
<option name="HIGHLIGHT_NON_ACTIVE_CHANGELIST" value="false" />
|
||||
<option name="LAST_RESOLUTION" value="IGNORE" />
|
||||
</component>
|
||||
<component name="CreatePatchCommitExecutor">
|
||||
<option name="PATCH_PATH" value="" />
|
||||
</component>
|
||||
<component name="ExecutionTargetManager" SELECTED_TARGET="default_target" />
|
||||
<component name="GradleLocalSettings">
|
||||
<option name="externalProjectsViewState">
|
||||
|
|
@ -22,8 +25,7 @@
|
|||
<sorting>DEFINITION_ORDER</sorting>
|
||||
</component>
|
||||
<component name="ProjectFrameBounds">
|
||||
<option name="x" value="34" />
|
||||
<option name="y" value="15" />
|
||||
<option name="y" value="23" />
|
||||
<option name="width" value="1346" />
|
||||
<option name="height" value="688" />
|
||||
</component>
|
||||
|
|
@ -42,9 +44,22 @@
|
|||
<foldersAlwaysOnTop value="true" />
|
||||
</navigator>
|
||||
<panes>
|
||||
<pane id="Scope" />
|
||||
<pane id="ProjectPane" />
|
||||
<pane id="Scratches" />
|
||||
<pane id="ProjectPane">
|
||||
<subPane>
|
||||
<PATH>
|
||||
<PATH_ELEMENT>
|
||||
<option name="myItemId" value="gather_program" />
|
||||
<option name="myItemType" value="com.intellij.ide.projectView.impl.nodes.ProjectViewProjectNode" />
|
||||
</PATH_ELEMENT>
|
||||
<PATH_ELEMENT>
|
||||
<option name="myItemId" value="gather_program" />
|
||||
<option name="myItemType" value="com.intellij.ide.projectView.impl.nodes.PsiDirectoryNode" />
|
||||
</PATH_ELEMENT>
|
||||
</PATH>
|
||||
</subPane>
|
||||
</pane>
|
||||
<pane id="Scope" />
|
||||
<pane id="PackagesPane" />
|
||||
</panes>
|
||||
</component>
|
||||
|
|
@ -169,37 +184,47 @@
|
|||
<option name="presentableId" value="Default" />
|
||||
<updated>1492270889131</updated>
|
||||
<workItem from="1492270892397" duration="475000" />
|
||||
<workItem from="1508007828124" duration="639000" />
|
||||
</task>
|
||||
<servers />
|
||||
</component>
|
||||
<component name="TimeTrackingManager">
|
||||
<option name="totallyTimeSpent" value="475000" />
|
||||
<option name="totallyTimeSpent" value="1114000" />
|
||||
</component>
|
||||
<component name="ToolWindowManager">
|
||||
<frame x="34" y="15" width="1346" height="688" extended-state="0" />
|
||||
<frame x="0" y="23" width="1346" height="688" extended-state="0" />
|
||||
<editor active="false" />
|
||||
<layout>
|
||||
<window_info id="Palette" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Palette" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="TODO" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="6" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Palette	" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Event Log" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="Maven Projects" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Version Control" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="false" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Nl-Palette" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Palette	" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Image Layers" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Capture Analysis" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Event Log" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="7" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="Maven Projects" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Run" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="2" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Terminal" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Designer" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Project" active="true" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" show_stripe_button="true" weight="0.24962406" sideWeight="0.5" order="0" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Database" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Version Control" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="false" weight="0.33" sideWeight="0.5" order="7" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Sequence" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Properties" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Spring" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Terminal" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="7" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Capture Tool" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Designer" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="2" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Project" active="true" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" show_stripe_button="true" weight="0.2530675" sideWeight="0.5" order="0" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Database" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Structure" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Ant Build" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="UI Designer" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Favorites" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="UI Designer" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="2" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Theme Preview" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="-1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Debug" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.4" sideWeight="0.5" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Favorites" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="2" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="Cvs" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="4" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Hierarchy" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="2" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Message" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="0" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Commander" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.4" sideWeight="0.5" order="0" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Find" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Inspection" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.4" sideWeight="0.5" order="5" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Hierarchy" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.25" sideWeight="0.5" order="2" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Find" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" show_stripe_button="true" weight="0.33" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
</layout>
|
||||
</component>
|
||||
<component name="TypeScriptGeneratedFilesManager">
|
||||
|
|
|
|||
Binary file not shown.
|
|
@ -0,0 +1,34 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<module org.jetbrains.idea.maven.project.MavenProjectsManager.isMavenModule="true" type="JAVA_MODULE" version="4">
|
||||
<component name="NewModuleRootManager" LANGUAGE_LEVEL="JDK_1_7" inherit-compiler-output="false">
|
||||
<output url="file://$MODULE_DIR$/target/classes" />
|
||||
<output-test url="file://$MODULE_DIR$/target/test-classes" />
|
||||
<content url="file://$MODULE_DIR$">
|
||||
<sourceFolder url="file://$MODULE_DIR$/src/main/java" isTestSource="false" />
|
||||
<excludeFolder url="file://$MODULE_DIR$/target" />
|
||||
</content>
|
||||
<orderEntry type="inheritedJdk" />
|
||||
<orderEntry type="sourceFolder" forTests="false" />
|
||||
<orderEntry type="library" scope="TEST" name="Maven: junit:junit:3.8.1" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.slf4j:slf4j-log4j12:1.7.7" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.slf4j:slf4j-api:1.7.7" level="project" />
|
||||
<orderEntry type="library" name="Maven: log4j:log4j:1.2.17" level="project" />
|
||||
<orderEntry type="library" name="Maven: commons-collections:commons-collections:3.2.1" level="project" />
|
||||
<orderEntry type="library" name="Maven: commons-io:commons-io:1.3.2" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.springframework:spring-context:4.1.4.RELEASE" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.springframework:spring-aop:4.1.4.RELEASE" level="project" />
|
||||
<orderEntry type="library" name="Maven: aopalliance:aopalliance:1.0" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.springframework:spring-beans:4.1.4.RELEASE" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.springframework:spring-core:4.1.4.RELEASE" level="project" />
|
||||
<orderEntry type="library" name="Maven: commons-logging:commons-logging:1.2" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.springframework:spring-expression:4.1.4.RELEASE" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.apache.commons:commons-lang3:3.1" level="project" />
|
||||
<orderEntry type="library" name="Maven: mysql:mysql-connector-java:5.1.18" level="project" />
|
||||
<orderEntry type="library" name="Maven: commons-dbcp:commons-dbcp:1.3" level="project" />
|
||||
<orderEntry type="library" name="Maven: commons-pool:commons-pool:1.5.4" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.mybatis:mybatis:3.1.1" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.mybatis:mybatis-spring:1.1.1" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.springframework:spring-tx:3.1.1.RELEASE" level="project" />
|
||||
<orderEntry type="library" name="Maven: org.springframework:spring-jdbc:3.1.1.RELEASE" level="project" />
|
||||
</component>
|
||||
</module>
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
|
|
@ -0,0 +1,16 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="CompilerConfiguration">
|
||||
<annotationProcessing>
|
||||
<profile name="Maven default annotation processors profile" enabled="true">
|
||||
<sourceOutputDir name="target/generated-sources/annotations" />
|
||||
<sourceTestOutputDir name="target/generated-test-sources/test-annotations" />
|
||||
<outputRelativeToContentRoot value="true" />
|
||||
<module name="match_program" />
|
||||
</profile>
|
||||
</annotationProcessing>
|
||||
<bytecodeTargetLevel>
|
||||
<module name="match_program" target="1.7" />
|
||||
</bytecodeTargetLevel>
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,6 @@
|
|||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project version="4">
|
||||
<component name="Encoding">
|
||||
<file url="file://$PROJECT_DIR$" charset="UTF-8" />
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: aopalliance:aopalliance:1.0">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/aopalliance/aopalliance/1.0/aopalliance-1.0.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/aopalliance/aopalliance/1.0/aopalliance-1.0-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/aopalliance/aopalliance/1.0/aopalliance-1.0-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: c3p0:c3p0:0.9.1.2">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/c3p0/c3p0/0.9.1.2/c3p0-0.9.1.2.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/c3p0/c3p0/0.9.1.2/c3p0-0.9.1.2-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/c3p0/c3p0/0.9.1.2/c3p0-0.9.1.2-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: commons-dbcp:commons-dbcp:1.3">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-dbcp/commons-dbcp/1.3/commons-dbcp-1.3.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-dbcp/commons-dbcp/1.3/commons-dbcp-1.3-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-dbcp/commons-dbcp/1.3/commons-dbcp-1.3-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: commons-logging:commons-logging:1.2">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-logging/commons-logging/1.2/commons-logging-1.2.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-logging/commons-logging/1.2/commons-logging-1.2-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-logging/commons-logging/1.2/commons-logging-1.2-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: commons-pool:commons-pool:1.5.4">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-pool/commons-pool/1.5.4/commons-pool-1.5.4.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-pool/commons-pool/1.5.4/commons-pool-1.5.4-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/commons-pool/commons-pool/1.5.4/commons-pool-1.5.4-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: log4j:log4j:1.2.17">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/log4j/log4j/1.2.17/log4j-1.2.17.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/log4j/log4j/1.2.17/log4j-1.2.17-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/log4j/log4j/1.2.17/log4j-1.2.17-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: mysql:mysql-connector-java:5.1.30">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/mysql/mysql-connector-java/5.1.30/mysql-connector-java-5.1.30.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/mysql/mysql-connector-java/5.1.30/mysql-connector-java-5.1.30-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/mysql/mysql-connector-java/5.1.30/mysql-connector-java-5.1.30-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.apache.lucene:lucene-analyzers-common:5.0.0">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/apache/lucene/lucene-analyzers-common/5.0.0/lucene-analyzers-common-5.0.0.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/apache/lucene/lucene-analyzers-common/5.0.0/lucene-analyzers-common-5.0.0-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/apache/lucene/lucene-analyzers-common/5.0.0/lucene-analyzers-common-5.0.0-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
|
|
@ -0,0 +1,13 @@
|
|||
<component name="libraryTable">
|
||||
<library name="Maven: org.apache.lucene:lucene-analyzers-smartcn:5.0.0">
|
||||
<CLASSES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/apache/lucene/lucene-analyzers-smartcn/5.0.0/lucene-analyzers-smartcn-5.0.0.jar!/" />
|
||||
</CLASSES>
|
||||
<JAVADOC>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/apache/lucene/lucene-analyzers-smartcn/5.0.0/lucene-analyzers-smartcn-5.0.0-javadoc.jar!/" />
|
||||
</JAVADOC>
|
||||
<SOURCES>
|
||||
<root url="jar://$MAVEN_REPOSITORY$/org/apache/lucene/lucene-analyzers-smartcn/5.0.0/lucene-analyzers-smartcn-5.0.0-sources.jar!/" />
|
||||
</SOURCES>
|
||||
</library>
|
||||
</component>
|
||||
Some files were not shown because too many files have changed in this diff Show More
Loading…
Reference in New Issue