parent
78a9d9ff79
commit
da0a0934f0
|
|
@ -1,4 +1,5 @@
|
|||
final/
|
||||
result/
|
||||
result*/
|
||||
result_backup/
|
||||
.idea
|
||||
|
|
@ -15,10 +15,12 @@
|
|||
<component name="ChangesViewManager" flattened_view="true" show_ignored="false" />
|
||||
<component name="CoverageDataManager">
|
||||
<SUITE FILE_PATH="coverage/word_process$classifier_for_yu.coverage" NAME="classifier_for_yu Coverage Results" MODIFIED="1453213349400" SOURCE_PROVIDER="com.intellij.coverage.DefaultCoverageFileProvider" RUNNER="coverage.py" COVERAGE_BY_TEST_ENABLED="true" COVERAGE_TRACING_ENABLED="false" WORKING_DIRECTORY="$PROJECT_DIR$" />
|
||||
<SUITE FILE_PATH="coverage/word_process$get_analysis_result.coverage" NAME="get_analysis_result Coverage Results" MODIFIED="1453220521239" SOURCE_PROVIDER="com.intellij.coverage.DefaultCoverageFileProvider" RUNNER="coverage.py" COVERAGE_BY_TEST_ENABLED="true" COVERAGE_TRACING_ENABLED="false" WORKING_DIRECTORY="$PROJECT_DIR$" />
|
||||
<SUITE FILE_PATH="coverage/word_process$get_analysis_result.coverage" NAME="get_analysis_result Coverage Results" MODIFIED="1453293026368" SOURCE_PROVIDER="com.intellij.coverage.DefaultCoverageFileProvider" RUNNER="coverage.py" COVERAGE_BY_TEST_ENABLED="true" COVERAGE_TRACING_ENABLED="false" WORKING_DIRECTORY="$PROJECT_DIR$" />
|
||||
<SUITE FILE_PATH="coverage/word_process$test.coverage" NAME="test Coverage Results" MODIFIED="1453298227598" SOURCE_PROVIDER="com.intellij.coverage.DefaultCoverageFileProvider" RUNNER="coverage.py" COVERAGE_BY_TEST_ENABLED="true" COVERAGE_TRACING_ENABLED="false" WORKING_DIRECTORY="$PROJECT_DIR$" />
|
||||
<SUITE FILE_PATH="coverage/word_process$get_feature_analysis_result.coverage" NAME="get_feature_analysis_result Coverage Results" MODIFIED="1453273038384" SOURCE_PROVIDER="com.intellij.coverage.DefaultCoverageFileProvider" RUNNER="coverage.py" COVERAGE_BY_TEST_ENABLED="true" COVERAGE_TRACING_ENABLED="false" WORKING_DIRECTORY="$PROJECT_DIR$" />
|
||||
<SUITE FILE_PATH="coverage/word_process$classifier.coverage" NAME="classifier Coverage Results" MODIFIED="1452693962507" SOURCE_PROVIDER="com.intellij.coverage.DefaultCoverageFileProvider" RUNNER="coverage.py" COVERAGE_BY_TEST_ENABLED="true" COVERAGE_TRACING_ENABLED="false" WORKING_DIRECTORY="$PROJECT_DIR$" />
|
||||
<SUITE FILE_PATH="coverage/word_process$sentence_classifier.coverage" NAME="sentence_classifier Coverage Results" MODIFIED="1452735271324" SOURCE_PROVIDER="com.intellij.coverage.DefaultCoverageFileProvider" RUNNER="coverage.py" COVERAGE_BY_TEST_ENABLED="true" COVERAGE_TRACING_ENABLED="false" WORKING_DIRECTORY="$PROJECT_DIR$" />
|
||||
<SUITE FILE_PATH="coverage/word_process$sentence_classifier.coverage" NAME="sentence_classifier Coverage Results" MODIFIED="1453280943351" SOURCE_PROVIDER="com.intellij.coverage.DefaultCoverageFileProvider" RUNNER="coverage.py" COVERAGE_BY_TEST_ENABLED="true" COVERAGE_TRACING_ENABLED="false" WORKING_DIRECTORY="$PROJECT_DIR$" />
|
||||
<SUITE FILE_PATH="coverage/word_process$sentence_classifier2.coverage" NAME="sentence_classifier2 Coverage Results" MODIFIED="1453295501125" SOURCE_PROVIDER="com.intellij.coverage.DefaultCoverageFileProvider" RUNNER="coverage.py" COVERAGE_BY_TEST_ENABLED="true" COVERAGE_TRACING_ENABLED="false" WORKING_DIRECTORY="$PROJECT_DIR$" />
|
||||
</component>
|
||||
<component name="CreatePatchCommitExecutor">
|
||||
<option name="PATCH_PATH" value="" />
|
||||
|
|
@ -33,7 +35,7 @@
|
|||
<entry file="file://$PROJECT_DIR$/get_analysis_result.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="142" column="73" selection-start-line="142" selection-start-column="73" selection-end-line="142" selection-end-column="73" />
|
||||
<caret line="192" column="0" selection-start-line="192" selection-start-column="0" selection-end-line="193" selection-end-column="0" />
|
||||
<folding>
|
||||
<element signature="e#0#13#0" expanded="true" />
|
||||
</folding>
|
||||
|
|
@ -45,7 +47,7 @@
|
|||
<entry file="file://$PROJECT_DIR$/classifier_for_yu.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="59" column="34" selection-start-line="59" selection-start-column="34" selection-end-line="59" selection-end-column="34" />
|
||||
<caret line="65" column="19" selection-start-line="49" selection-start-column="4" selection-end-line="65" selection-end-column="19" />
|
||||
<folding>
|
||||
<element signature="e#0#23#0" expanded="true" />
|
||||
</folding>
|
||||
|
|
@ -53,11 +55,23 @@
|
|||
</provider>
|
||||
</entry>
|
||||
</file>
|
||||
<file leaf-file-name="get_feature_analysis_result.py" pinned="false" current-in-tab="false">
|
||||
<entry file="file://$PROJECT_DIR$/get_feature_analysis_result.py">
|
||||
<file leaf-file-name="sentence_classifier.py" pinned="false" current-in-tab="false">
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="236" column="18" selection-start-line="236" selection-start-column="18" selection-end-line="236" selection-end-column="18" />
|
||||
<caret line="10" column="0" selection-start-line="10" selection-start-column="0" selection-end-line="10" selection-end-column="0" />
|
||||
<folding>
|
||||
<element signature="e#19#30#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
</file>
|
||||
<file leaf-file-name="get_feature_analysis_result.py" pinned="false" current-in-tab="true">
|
||||
<entry file="file://$PROJECT_DIR$/get_feature_analysis_result.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.8074324">
|
||||
<caret line="258" column="0" selection-start-line="258" selection-start-column="0" selection-end-line="258" selection-end-column="0" />
|
||||
<folding>
|
||||
<element signature="e#0#13#0" expanded="true" />
|
||||
</folding>
|
||||
|
|
@ -65,13 +79,13 @@
|
|||
</provider>
|
||||
</entry>
|
||||
</file>
|
||||
<file leaf-file-name="sentence_classifier.py" pinned="false" current-in-tab="true">
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier.py">
|
||||
<file leaf-file-name="classifier.py" pinned="false" current-in-tab="false">
|
||||
<entry file="file://$PROJECT_DIR$/classifier.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.23131673">
|
||||
<caret line="217" column="44" selection-start-line="217" selection-start-column="44" selection-end-line="217" selection-end-column="44" />
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="0" column="0" selection-start-line="0" selection-start-column="0" selection-end-line="0" selection-end-column="76" />
|
||||
<folding>
|
||||
<element signature="e#19#30#0" expanded="true" />
|
||||
<element signature="e#0#76#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
|
|
@ -81,7 +95,7 @@
|
|||
<entry file="file://$PROJECT_DIR$/helper.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="12" column="41" selection-start-line="12" selection-start-column="41" selection-end-line="12" selection-end-column="41" />
|
||||
<caret line="116" column="4" selection-start-line="116" selection-start-column="4" selection-end-line="116" selection-end-column="4" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
|
|
@ -104,17 +118,36 @@
|
|||
</provider>
|
||||
</entry>
|
||||
</file>
|
||||
<file leaf-file-name="sentence_classifier2.py" pinned="false" current-in-tab="false">
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier2.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="311" column="23" selection-start-line="311" selection-start-column="23" selection-end-line="311" selection-end-column="75" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
</file>
|
||||
</leaf>
|
||||
</component>
|
||||
<component name="FileTemplateManagerImpl">
|
||||
<option name="RECENT_TEMPLATES">
|
||||
<list>
|
||||
<option value="Python Script" />
|
||||
</list>
|
||||
</option>
|
||||
</component>
|
||||
<component name="IdeDocumentHistory">
|
||||
<option name="CHANGED_PATHS">
|
||||
<list>
|
||||
<option value="$PROJECT_DIR$/classifier.py" />
|
||||
<option value="$PROJECT_DIR$/sentence_classifier.py" />
|
||||
<option value="$PROJECT_DIR$/dao.py" />
|
||||
<option value="$PROJECT_DIR$/classifier_for_yu.py" />
|
||||
<option value="$PROJECT_DIR$/get_analysis_result.py" />
|
||||
<option value="$PROJECT_DIR$/get_feature_analysis_result.py" />
|
||||
<option value="$PROJECT_DIR$/get_analysis_result.py" />
|
||||
<option value="$PROJECT_DIR$/classifier.py" />
|
||||
<option value="$PROJECT_DIR$/sentence_classifier.py" />
|
||||
<option value="$PROJECT_DIR$/sentence_classifier2.py" />
|
||||
<option value="$PROJECT_DIR$/test.py" />
|
||||
</list>
|
||||
</option>
|
||||
</component>
|
||||
|
|
@ -154,6 +187,8 @@
|
|||
<sortByType />
|
||||
</navigator>
|
||||
<panes>
|
||||
<pane id="Scratches" />
|
||||
<pane id="Scope" />
|
||||
<pane id="ProjectPane">
|
||||
<subPane>
|
||||
<PATH>
|
||||
|
|
@ -174,8 +209,6 @@
|
|||
</PATH>
|
||||
</subPane>
|
||||
</pane>
|
||||
<pane id="Scope" />
|
||||
<pane id="Scratches" />
|
||||
</panes>
|
||||
</component>
|
||||
<component name="PropertiesComponent">
|
||||
|
|
@ -186,7 +219,7 @@
|
|||
<property name="settings.editor.selected.configurable" value="reference.settingsdialog.IDE.editor.colors.Font" />
|
||||
<property name="settings.editor.splitter.proportion" value="0.2" />
|
||||
</component>
|
||||
<component name="RunManager" selected="Python.get_feature_analysis_result">
|
||||
<component name="RunManager" selected="Python.test">
|
||||
<configuration default="false" name="sentence_classifier" type="PythonConfigurationType" factoryName="Python" temporary="true">
|
||||
<option name="INTERPRETER_OPTIONS" value="" />
|
||||
<option name="PARENT_ENVS" value="true" />
|
||||
|
|
@ -205,24 +238,6 @@
|
|||
<option name="SHOW_COMMAND_LINE" value="false" />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="false" name="classifier" type="PythonConfigurationType" factoryName="Python" temporary="true">
|
||||
<option name="INTERPRETER_OPTIONS" value="" />
|
||||
<option name="PARENT_ENVS" value="true" />
|
||||
<envs>
|
||||
<env name="PYTHONUNBUFFERED" value="1" />
|
||||
</envs>
|
||||
<option name="SDK_HOME" value="" />
|
||||
<option name="WORKING_DIRECTORY" value="$PROJECT_DIR$" />
|
||||
<option name="IS_MODULE_SDK" value="true" />
|
||||
<option name="ADD_CONTENT_ROOTS" value="true" />
|
||||
<option name="ADD_SOURCE_ROOTS" value="true" />
|
||||
<module name="word_process" />
|
||||
<EXTENSION ID="PythonCoverageRunConfigurationExtension" enabled="false" sample_coverage="true" runner="coverage.py" />
|
||||
<option name="SCRIPT_NAME" value="$PROJECT_DIR$/classifier.py" />
|
||||
<option name="PARAMETERS" value="" />
|
||||
<option name="SHOW_COMMAND_LINE" value="false" />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="false" name="get_analysis_result" type="PythonConfigurationType" factoryName="Python" temporary="true">
|
||||
<option name="INTERPRETER_OPTIONS" value="" />
|
||||
<option name="PARENT_ENVS" value="true" />
|
||||
|
|
@ -259,7 +274,7 @@
|
|||
<option name="SHOW_COMMAND_LINE" value="false" />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="false" name="classifier_for_yu" type="PythonConfigurationType" factoryName="Python" temporary="true">
|
||||
<configuration default="false" name="sentence_classifier2" type="PythonConfigurationType" factoryName="Python" temporary="true">
|
||||
<option name="INTERPRETER_OPTIONS" value="" />
|
||||
<option name="PARENT_ENVS" value="true" />
|
||||
<envs>
|
||||
|
|
@ -272,7 +287,25 @@
|
|||
<option name="ADD_SOURCE_ROOTS" value="true" />
|
||||
<module name="word_process" />
|
||||
<EXTENSION ID="PythonCoverageRunConfigurationExtension" enabled="false" sample_coverage="true" runner="coverage.py" />
|
||||
<option name="SCRIPT_NAME" value="$PROJECT_DIR$/classifier_for_yu.py" />
|
||||
<option name="SCRIPT_NAME" value="$PROJECT_DIR$/sentence_classifier2.py" />
|
||||
<option name="PARAMETERS" value="" />
|
||||
<option name="SHOW_COMMAND_LINE" value="false" />
|
||||
<method />
|
||||
</configuration>
|
||||
<configuration default="false" name="test" type="PythonConfigurationType" factoryName="Python" temporary="true">
|
||||
<option name="INTERPRETER_OPTIONS" value="" />
|
||||
<option name="PARENT_ENVS" value="true" />
|
||||
<envs>
|
||||
<env name="PYTHONUNBUFFERED" value="1" />
|
||||
</envs>
|
||||
<option name="SDK_HOME" value="" />
|
||||
<option name="WORKING_DIRECTORY" value="$PROJECT_DIR$" />
|
||||
<option name="IS_MODULE_SDK" value="true" />
|
||||
<option name="ADD_CONTENT_ROOTS" value="true" />
|
||||
<option name="ADD_SOURCE_ROOTS" value="true" />
|
||||
<module name="word_process" />
|
||||
<EXTENSION ID="PythonCoverageRunConfigurationExtension" enabled="false" sample_coverage="true" runner="coverage.py" />
|
||||
<option name="SCRIPT_NAME" value="$PROJECT_DIR$/test.py" />
|
||||
<option name="PARAMETERS" value="" />
|
||||
<option name="SHOW_COMMAND_LINE" value="false" />
|
||||
<method />
|
||||
|
|
@ -467,18 +500,18 @@
|
|||
</configuration>
|
||||
<list size="5">
|
||||
<item index="0" class="java.lang.String" itemvalue="Python.sentence_classifier" />
|
||||
<item index="1" class="java.lang.String" itemvalue="Python.classifier" />
|
||||
<item index="2" class="java.lang.String" itemvalue="Python.get_analysis_result" />
|
||||
<item index="3" class="java.lang.String" itemvalue="Python.get_feature_analysis_result" />
|
||||
<item index="4" class="java.lang.String" itemvalue="Python.classifier_for_yu" />
|
||||
<item index="1" class="java.lang.String" itemvalue="Python.get_analysis_result" />
|
||||
<item index="2" class="java.lang.String" itemvalue="Python.get_feature_analysis_result" />
|
||||
<item index="3" class="java.lang.String" itemvalue="Python.sentence_classifier2" />
|
||||
<item index="4" class="java.lang.String" itemvalue="Python.test" />
|
||||
</list>
|
||||
<recent_temporary>
|
||||
<list size="5">
|
||||
<item index="0" class="java.lang.String" itemvalue="Python.get_feature_analysis_result" />
|
||||
<item index="1" class="java.lang.String" itemvalue="Python.get_analysis_result" />
|
||||
<item index="2" class="java.lang.String" itemvalue="Python.classifier_for_yu" />
|
||||
<item index="0" class="java.lang.String" itemvalue="Python.test" />
|
||||
<item index="1" class="java.lang.String" itemvalue="Python.sentence_classifier2" />
|
||||
<item index="2" class="java.lang.String" itemvalue="Python.get_analysis_result" />
|
||||
<item index="3" class="java.lang.String" itemvalue="Python.sentence_classifier" />
|
||||
<item index="4" class="java.lang.String" itemvalue="Python.classifier" />
|
||||
<item index="4" class="java.lang.String" itemvalue="Python.get_feature_analysis_result" />
|
||||
</list>
|
||||
</recent_temporary>
|
||||
</component>
|
||||
|
|
@ -496,7 +529,7 @@
|
|||
<servers />
|
||||
</component>
|
||||
<component name="ToolWindowManager">
|
||||
<frame x="-8" y="-8" width="1936" height="1056" extended-state="7" />
|
||||
<frame x="-8" y="-8" width="1936" height="1056" extended-state="6" />
|
||||
<editor active="true" />
|
||||
<layout>
|
||||
<window_info id="TODO" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="6" side_tool="false" content_ui="tabs" />
|
||||
|
|
@ -506,12 +539,12 @@
|
|||
<window_info id="Version Control" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="7" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Run" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" weight="0.32792208" sideWeight="0.47441363" order="2" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Terminal" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.32900432" sideWeight="0.48507464" order="7" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Project" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" weight="0.15511727" sideWeight="0.5" order="0" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Project" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" weight="0.15671642" sideWeight="0.5" order="0" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Database" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Find" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Structure" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.25" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Favorites" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="2" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="Debug" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.4" sideWeight="0.7677039" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Debug" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.39935064" sideWeight="0.7677039" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Cvs" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.25" sideWeight="0.5" order="4" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Message" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="0" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Commander" active="false" anchor="right" auto_hide="false" internal_type="SLIDING" type="SLIDING" visible="false" weight="0.4" sideWeight="0.5" order="0" side_tool="false" content_ui="tabs" />
|
||||
|
|
@ -533,15 +566,92 @@
|
|||
</component>
|
||||
<component name="XDebuggerManager">
|
||||
<breakpoint-manager>
|
||||
<option name="time" value="1" />
|
||||
<option name="time" value="4" />
|
||||
</breakpoint-manager>
|
||||
<watches-manager>
|
||||
<configuration name="PythonConfigurationType">
|
||||
<watch expression="data" language="Python" />
|
||||
<watch expression="kf" language="Python" />
|
||||
<watch expression="kf[0]" language="Python" />
|
||||
<watch expression="kf.idxs" language="Python" />
|
||||
<watch expression="X_train" language="Python" />
|
||||
<watch expression="train_index" language="Python" />
|
||||
<watch expression="test_index" language="Python" />
|
||||
</configuration>
|
||||
</watches-manager>
|
||||
</component>
|
||||
<component name="editorHistoryManager">
|
||||
<entry file="file://$PROJECT_DIR$/get_analysis_result.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="0" column="0" selection-start-line="0" selection-start-column="0" selection-end-line="0" selection-end-column="0" />
|
||||
<folding>
|
||||
<element signature="e#0#13#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/classifier_for_yu.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="59" column="34" selection-start-line="59" selection-start-column="34" selection-end-line="59" selection-end-column="34" />
|
||||
<folding>
|
||||
<element signature="e#0#23#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="10" column="0" selection-start-line="10" selection-start-column="0" selection-end-line="10" selection-end-column="0" />
|
||||
<folding>
|
||||
<element signature="e#19#30#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/get_feature_analysis_result.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="236" column="18" selection-start-line="236" selection-start-column="18" selection-end-line="236" selection-end-column="18" />
|
||||
<folding>
|
||||
<element signature="e#0#13#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier2.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="22" column="38" selection-start-line="22" selection-start-column="38" selection-end-line="22" selection-end-column="38" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/helper.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="116" column="4" selection-start-line="116" selection-start-column="4" selection-end-line="116" selection-end-column="4" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/dao.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="54" column="17" selection-start-line="54" selection-start-column="17" selection-end-line="54" selection-end-column="17" />
|
||||
<folding>
|
||||
<marker date="1453213345765" expanded="true" signature="252:294" placeholder="select title..issues..." />
|
||||
<marker date="1453213345765" expanded="true" signature="470:608" placeholder="select issue..issues..." />
|
||||
<marker date="1453213345765" expanded="true" signature="732:859" placeholder="select proje..issues..." />
|
||||
<marker date="1453213345765" expanded="true" signature="1170:1285" placeholder="select title..issues_for_yu..." />
|
||||
<marker date="1453213345765" expanded="true" signature="1513:1551" placeholder="update issue..." />
|
||||
<marker date="1453213345765" expanded="true" signature="1513:1587" placeholder="update issue..." />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/get_analysis_result.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
|
|
@ -638,13 +748,6 @@
|
|||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/classifier.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.1285956">
|
||||
<caret line="5" column="0" selection-start-line="5" selection-start-column="0" selection-end-line="5" selection-end-column="0" />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://E:/Program Files/Canopy/User/Lib/site-packages/pymysql/err.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.5177665">
|
||||
|
|
@ -652,14 +755,6 @@
|
|||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/helper.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="12" column="41" selection-start-line="12" selection-start-column="41" selection-end-line="12" selection-end-column="41" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/dao.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
|
|
@ -675,12 +770,20 @@
|
|||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/classifier_for_yu.py">
|
||||
<entry file="file://$PROJECT_DIR$/helper.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="59" column="34" selection-start-line="59" selection-start-column="34" selection-end-line="59" selection-end-column="34" />
|
||||
<caret line="116" column="4" selection-start-line="116" selection-start-column="4" selection-end-line="116" selection-end-column="4" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="10" column="0" selection-start-line="10" selection-start-column="0" selection-end-line="10" selection-end-column="0" />
|
||||
<folding>
|
||||
<element signature="e#0#23#0" expanded="true" />
|
||||
<element signature="e#19#30#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
|
|
@ -688,32 +791,58 @@
|
|||
<entry file="file://$PROJECT_DIR$/get_analysis_result.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="142" column="73" selection-start-line="142" selection-start-column="73" selection-end-line="142" selection-end-column="73" />
|
||||
<caret line="192" column="0" selection-start-line="192" selection-start-column="0" selection-end-line="193" selection-end-column="0" />
|
||||
<folding>
|
||||
<element signature="e#0#13#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/classifier_for_yu.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="65" column="19" selection-start-line="49" selection-start-column="4" selection-end-line="65" selection-end-column="19" />
|
||||
<folding>
|
||||
<element signature="e#0#23#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/classifier.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="0" column="0" selection-start-line="0" selection-start-column="0" selection-end-line="0" selection-end-column="76" />
|
||||
<folding>
|
||||
<element signature="e#0#76#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier2.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="311" column="23" selection-start-line="311" selection-start-column="23" selection-end-line="311" selection-end-column="75" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/test.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="0" column="0" selection-start-line="0" selection-start-column="0" selection-end-line="0" selection-end-column="0" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/get_feature_analysis_result.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="236" column="18" selection-start-line="236" selection-start-column="18" selection-end-line="236" selection-end-column="18" />
|
||||
<state vertical-scroll-proportion="0.8074324">
|
||||
<caret line="258" column="0" selection-start-line="258" selection-start-column="0" selection-end-line="258" selection-end-column="0" />
|
||||
<folding>
|
||||
<element signature="e#0#13#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.23131673">
|
||||
<caret line="217" column="44" selection-start-line="217" selection-start-column="44" selection-end-line="217" selection-end-column="44" />
|
||||
<folding>
|
||||
<element signature="e#19#30#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
</component>
|
||||
</project>
|
||||
|
|
@ -31,7 +31,7 @@ def data_preprocess(project_name, methold_name = ''):
|
|||
train_target = []
|
||||
x_id = []
|
||||
for r in cur.fetchall():
|
||||
train_data.append(helper.filter_str(repr(r[1])+"\n"+repr(r[2])))
|
||||
train_data.append(helper.filter_str(repr(r[1])+".\n"+repr(r[2])))
|
||||
train_target.append(r[0])
|
||||
x_id.append(r[3])
|
||||
print("data length is : ", len(train_target))
|
||||
|
|
|
|||
BIN
classifier.pyc
BIN
classifier.pyc
Binary file not shown.
|
|
@ -149,22 +149,15 @@ def analysis_process(project_id,method):
|
|||
return count_improve*1.0/int(result_set[0][-1]),r1,\
|
||||
result_set[0][-1],result_set[0][final_threshold]
|
||||
|
||||
|
||||
method = 'svm'
|
||||
final_path = 'final/'
|
||||
helper.mkdir(final_path)
|
||||
f_proj_id = file('proj_id.csv', 'r')
|
||||
reader = csv.reader(f_proj_id)
|
||||
f_all_precision = file(final_path + 'precision_'+method+'.csv','w')
|
||||
writer_final = csv.writer(f_all_precision)
|
||||
precision_improve = []
|
||||
precision_machine = []
|
||||
p_id = []
|
||||
writer_final.writerow(['proj_id','issue_count','proj_name','improve_prec','improve_prec_part','mechine_prec','num_sample','num_hard'])
|
||||
|
||||
line = ['6','500']
|
||||
for line in reader:
|
||||
# classifier_process(line[0])
|
||||
def handle_project(line):
|
||||
method = 'svm'
|
||||
final_path = 'final/'
|
||||
helper.mkdir(final_path)
|
||||
f_all_precision = file(final_path + 'precision_'+method+'.csv','a')
|
||||
writer_final = csv.writer(f_all_precision)
|
||||
precision_improve = []
|
||||
precision_machine = []
|
||||
p_id = []
|
||||
|
||||
print(line[0])
|
||||
p_id.append(line[0])
|
||||
|
|
@ -186,8 +179,22 @@ for line in reader:
|
|||
# data = (line[0],line[1],proj_name,analysis_process(line[0]),helper.get_precision(line[0]),et,et_1000,rf,nb,lrl1,lrl2,adaboost)
|
||||
data = (line[0],line[1],proj_name,temp[0],temp[1],helper.get_precision(line[0],method),temp[2],temp[3])
|
||||
writer_final.writerow(data)
|
||||
# classifier_process(line[0])
|
||||
f_all_precision.close()
|
||||
|
||||
|
||||
|
||||
f_proj_id = file('proj_id.csv', 'r')
|
||||
reader = csv.reader(f_proj_id)
|
||||
# line = ['6','500']
|
||||
method = 'svm'
|
||||
final_path = 'final/'
|
||||
helper.mkdir(final_path)
|
||||
f_all_precision = file(final_path + 'precision_'+method+'.csv','a')
|
||||
writer_final = csv.writer(f_all_precision)
|
||||
writer_final.writerow(['proj_id','issue_count','proj_name','improve_prec','improve_prec_part','mechine_prec','num_sample','num_hard'])
|
||||
f_all_precision.close()
|
||||
for line in reader:
|
||||
writer_final = csv.writer(f_all_precision)
|
||||
|
||||
handle_project(line)
|
||||
|
||||
f_proj_id.close()
|
||||
|
|
@ -0,0 +1,5 @@
|
|||
classifier.py:对数据进行预处理,包括stemming,去除数字,特殊字符,TFIDF计算
|
||||
sentence_classifier.py:分类程序,包括直接使用svm,对hard部分的特殊处理,结果统计与持久化
|
||||
dao:数据库访问方法
|
||||
helper:一些复杂方法的实现
|
||||
get_analysis_result/get_feature_analysis_result:对分类结果的分析
|
||||
|
|
@ -7,7 +7,7 @@ from sklearn.cross_validation import KFold
|
|||
from time import time
|
||||
import csv
|
||||
# from itertools import *
|
||||
# import dao
|
||||
import dao
|
||||
|
||||
import sys
|
||||
import helper
|
||||
|
|
|
|||
|
|
@ -0,0 +1,429 @@
|
|||
# import traceback
|
||||
import nltk
|
||||
import numpy as np
|
||||
from sklearn import svm
|
||||
from sklearn import metrics
|
||||
from sklearn.cross_validation import KFold
|
||||
from time import time
|
||||
import csv
|
||||
# from itertools import *
|
||||
import dao
|
||||
|
||||
import sys
|
||||
import helper
|
||||
import classifier
|
||||
|
||||
reload(sys)
|
||||
sys.setdefaultencoding("utf-8")
|
||||
|
||||
__author__ = 'mac'
|
||||
import cPickle as pickle
|
||||
|
||||
|
||||
def classifier_project_by_id(proj_id):
|
||||
# project_name = repr(proj_id)
|
||||
project_name = proj_id
|
||||
methold_name = 'svm'
|
||||
|
||||
path = 'result/'+project_name+'/'
|
||||
helper.mkdir(path)
|
||||
path_analysis = path + methold_name + '/analysis/'
|
||||
helper.mkdir(path_analysis)
|
||||
path_data = path + methold_name + '/data/'
|
||||
helper.mkdir(path_data)
|
||||
csv_result = file(path_analysis + 'precision_method.csv', 'wb')
|
||||
writer_result = csv.writer(csv_result)
|
||||
|
||||
feature_csv_result = file(path_analysis + 'feature_precision_method.csv', 'wb')
|
||||
feature_writer_result = csv.writer(feature_csv_result)
|
||||
|
||||
csv_path = file(path_analysis + 'sentence_split.csv', 'wb')
|
||||
writer = csv.writer(csv_path)
|
||||
writer.writerow(['id','diff','y_test','pred','sentence_pred','sentence_pred_0','sentence_pred_1','sentence'])
|
||||
|
||||
|
||||
csv_classifier = file(path_analysis + 'classifier_info.csv', 'wb')
|
||||
writer_classifier = csv.writer(csv_classifier)
|
||||
writer_classifier.writerow(['classifier infomation for project:',repr(project_name)])
|
||||
|
||||
threshold = []
|
||||
break_count = 20 # cut the different, len(threshold)
|
||||
change_break = 20 # cut the condition for changing class
|
||||
|
||||
for ind_c_th in range(1,break_count+1):
|
||||
threshold.append(ind_c_th*1.0/break_count)
|
||||
|
||||
count_base = [0]*break_count
|
||||
right_base = [0]*break_count
|
||||
count_base2 = [0]*break_count
|
||||
right_base2 = [0]*break_count
|
||||
|
||||
feature_count_base = [0]*break_count
|
||||
feature_pred_base = [0]*break_count
|
||||
feature_right_base = [0]*break_count
|
||||
feature_count_base2 = [0]*break_count
|
||||
feature_pred_base2 = [0]*break_count
|
||||
feature_right_base2 = [0]*break_count
|
||||
|
||||
|
||||
right_mine1 = [([0] * change_break) for i in range(break_count)]
|
||||
right_mine2 = [([0] * change_break) for i in range(break_count)]
|
||||
right_mine3 = [([0] * change_break) for i in range(break_count)]
|
||||
right_mine4 = [([0] * change_break) for i in range(break_count)]
|
||||
|
||||
feature_right_mine1 = [([0] * change_break) for i in range(break_count)]
|
||||
feature_right_mine2 = [([0] * change_break) for i in range(break_count)]
|
||||
feature_right_mine3 = [([0] * change_break) for i in range(break_count)]
|
||||
feature_right_mine4 = [([0] * change_break) for i in range(break_count)]
|
||||
|
||||
feature_count_mine1 = [([0] * change_break) for i in range(break_count)]
|
||||
feature_count_mine2 = [([0] * change_break) for i in range(break_count)]
|
||||
feature_count_mine3 = [([0] * change_break) for i in range(break_count)]
|
||||
feature_count_mine4 = [([0] * change_break) for i in range(break_count)]
|
||||
|
||||
|
||||
X,y,x_id_before,vect = classifier.data_preprocess(project_name,methold_name) # first time run, get tf-idf of train data
|
||||
# X,y,x_id_before,vect = helper.get_tfidf_data(project_name) # fellow methold run, to get tf-idf store in disk
|
||||
f_train_data = path + 'train_data.pkl'
|
||||
train_data = helper.get_pickle_record(f_train_data)
|
||||
|
||||
|
||||
# get tf-idf matirx of train data
|
||||
# print('get data: ')
|
||||
# f_train = path +'train.pkl'
|
||||
# f_target = path + 'target.pkl'
|
||||
# f_id = path + 'id.pkl'
|
||||
# f_vect = path + 'vect.pkl'
|
||||
# X = helper.get_pickle_record(f_train)
|
||||
# y = helper.get_pickle_record(f_target)
|
||||
# x_id = helper.get_pickle_record(f_id)
|
||||
# vect = helper.get_pickle_record(f_vect)
|
||||
# print('done')
|
||||
|
||||
y = np.array(y)
|
||||
x_id = np.array(x_id_before)
|
||||
|
||||
results = []
|
||||
kf = KFold(len(y), n_folds=10)
|
||||
|
||||
|
||||
def change_flag(flag,ind_threshold,ind_change,j,sentences_count):
|
||||
flag_t = False
|
||||
if j == 0 or j == 1 or j == sentences_count:
|
||||
flag[ind_threshold][ind_change]=True
|
||||
else:
|
||||
# todo: according to possision to decide whether to change flag
|
||||
if sentences_count >= 4 and (j-1 <=1 or sentences_count-j <=1):
|
||||
flag_t = True
|
||||
if sentences_count >= 4 and (abs((j-1)*1.0/(sentences_count-j)) <= 1.0/3 or abs((sentences_count-j)*1.0/(j-1)) <= 1.0/3):
|
||||
flag_t = True
|
||||
if flag_t:
|
||||
flag[ind_threshold][ind_change]=True
|
||||
|
||||
def data_process(change_flag_set,ind_threshold,status):
|
||||
if status:
|
||||
for ind_change in range(change_break):
|
||||
if not change_flag_set[0][ind_threshold][ind_change]:
|
||||
right_mine1[ind_threshold][ind_change] = right_mine1[ind_threshold][ind_change] + 1
|
||||
if not change_flag_set[1][ind_threshold][ind_change]:
|
||||
right_mine2[ind_threshold][ind_change] = right_mine2[ind_threshold][ind_change] + 1
|
||||
if not change_flag_set[2][ind_threshold][ind_change]:
|
||||
right_mine3[ind_threshold][ind_change] = right_mine3[ind_threshold][ind_change] + 1
|
||||
if not change_flag_set[3][ind_threshold][ind_change]:
|
||||
right_mine4[ind_threshold][ind_change] = right_mine4[ind_threshold][ind_change] + 1
|
||||
if not status:
|
||||
for ind_change in range(change_break):
|
||||
if change_flag_set[0][ind_threshold][ind_change]:
|
||||
right_mine1[ind_threshold][ind_change] = right_mine1[ind_threshold][ind_change] + 1
|
||||
if change_flag_set[1][ind_threshold][ind_change]:
|
||||
right_mine2[ind_threshold][ind_change] = right_mine2[ind_threshold][ind_change] + 1
|
||||
if change_flag_set[2][ind_threshold][ind_change]:
|
||||
right_mine3[ind_threshold][ind_change] = right_mine3[ind_threshold][ind_change] + 1
|
||||
if change_flag_set[3][ind_threshold][ind_change]:
|
||||
right_mine4[ind_threshold][ind_change] = right_mine4[ind_threshold][ind_change] + 1
|
||||
|
||||
def data_process2(change_flag_set,ind_threshold,status):
|
||||
for ind_change in range(change_break):
|
||||
if change_flag_set[0][ind_threshold][ind_change] and not status:
|
||||
feature_right_mine1[ind_threshold][ind_change] = feature_right_mine1[ind_threshold][ind_change] + 1
|
||||
if change_flag_set[1][ind_threshold][ind_change] and not status:
|
||||
feature_right_mine2[ind_threshold][ind_change] = feature_right_mine2[ind_threshold][ind_change] + 1
|
||||
if change_flag_set[2][ind_threshold][ind_change] and not status:
|
||||
feature_right_mine3[ind_threshold][ind_change] = feature_right_mine3[ind_threshold][ind_change] + 1
|
||||
if change_flag_set[3][ind_threshold][ind_change] and not status:
|
||||
feature_right_mine4[ind_threshold][ind_change] = feature_right_mine4[ind_threshold][ind_change] + 1
|
||||
|
||||
|
||||
turn_count = 0
|
||||
print("start training:")
|
||||
# ten-fold traning
|
||||
for train_index, test_index in kf:
|
||||
X_train, X_test = X[train_index], X[test_index]
|
||||
y_train, y_test = y[train_index], y[test_index]
|
||||
x_id_train, x_id_test = x_id[train_index], x_id[test_index]
|
||||
turn_count = turn_count+1
|
||||
print('turn '+ repr(turn_count) +':')
|
||||
writer_classifier.writerow(['------------------------------'])
|
||||
writer_classifier.writerow(['turn ', repr(turn_count) ,':'])
|
||||
# print('='*80)
|
||||
###############################################################################
|
||||
# Benchmark classifiers
|
||||
def benchmark(clf):
|
||||
print('_' * 80)
|
||||
print("Training: ")
|
||||
print(clf)
|
||||
t0 = time()
|
||||
clf.fit(X_train, y_train)
|
||||
train_time = time() - t0
|
||||
print("train time: %0.3fs" % train_time)
|
||||
writer_classifier.writerow(["train time:", train_time])
|
||||
|
||||
t0 = time()
|
||||
pred = clf.predict(X_test)
|
||||
|
||||
test_time = time() - t0
|
||||
print("test time: %0.3fs" % test_time)
|
||||
writer_classifier.writerow(["test time:", test_time])
|
||||
|
||||
score = metrics.accuracy_score(y_test, pred)
|
||||
print("accuracy: %0.3f" % score)
|
||||
writer_classifier.writerow(["accuracy:", score])
|
||||
|
||||
probability = clf.predict_proba(X_test)
|
||||
|
||||
f_mechine = path_data + 'mechine_' + repr(turn_count)
|
||||
with open(f_mechine, 'w') as f:
|
||||
pickle.dump(clf, f)
|
||||
|
||||
|
||||
np.save(path_data + "y_test_"+repr(turn_count),y_test)
|
||||
np.save(path_data + "pred_"+repr(turn_count),pred)
|
||||
np.save(path_data + "x_id_"+repr(turn_count),x_id_test)
|
||||
np.save(path_data + "probability_"+repr(turn_count),probability)
|
||||
|
||||
tokenizer = nltk.data.load('tokenizers/punkt/english.pickle')
|
||||
|
||||
for ind in range(len(pred)):
|
||||
# reset flag for each test data
|
||||
change_flag1 = [([False] * change_break) for i in range(break_count)]
|
||||
change_flag2 = [([False] * change_break) for i in range(break_count)]
|
||||
change_flag3 = [([False] * change_break) for i in range(break_count)]
|
||||
change_flag4 = [([False] * change_break) for i in range(break_count)]
|
||||
change_flag_set = [change_flag1,change_flag2,change_flag3,change_flag4]
|
||||
|
||||
# diff = np.sort(probability[ind:ind+1])[:1,-1:][0][0]-np.sort(probability[ind:ind+1],)[:1,-2:-1][0][0]
|
||||
# get sentence info and split it
|
||||
# issue = dao.get_info_by_id(x_id_test[ind])
|
||||
# for temp in issue:
|
||||
# issue_title = temp[0]
|
||||
# issue_body = temp[1]
|
||||
info = helper.get_info_by_id(x_id_test[ind],x_id_before,train_data)
|
||||
# issue_title = issue[1]
|
||||
# issue_body = issue[2]
|
||||
# info = issue_title + '.\n' + issue_body
|
||||
# info = ''.join(ifilterfalse(unicode.isdigit, info))
|
||||
info = helper.filter_str(info)
|
||||
|
||||
sentences = tokenizer.tokenize(info)
|
||||
x_test = vect.transform(sentences)
|
||||
x_pred = clf.predict(x_test)
|
||||
x_prob = clf.predict_proba(x_test)
|
||||
diff = probability[ind:ind+1,0][0]-probability[ind:ind+1,1][0]
|
||||
|
||||
# get num of sentences for dividing body (without title)
|
||||
sentences_count = len(x_pred)-1
|
||||
# record split information
|
||||
if len(x_pred)>0:
|
||||
for j in range(len(x_pred)):
|
||||
if(helper.word_count(sentences[j])>3):
|
||||
# data format : ['id','diff','y_test','pred','sentence_pred','sentence_pred_0','sentence_pred_1','sentence']
|
||||
data = (x_id_test[ind],diff,y_test[ind],pred[ind],x_pred[j],x_prob[j:j+1,0][0],x_prob[j:j+1,1][0],sentences[j])
|
||||
|
||||
writer.writerow(data)
|
||||
|
||||
for ind_threshold in range(break_count):
|
||||
if abs(diff) <= threshold[ind_threshold]:
|
||||
count_base[ind_threshold] = count_base[ind_threshold] + 1
|
||||
if int(y_test[ind]) == 1:
|
||||
feature_count_base[ind_threshold] = feature_count_base[ind_threshold] +1
|
||||
if int(pred[ind]) == 1:
|
||||
feature_pred_base[ind_threshold] = feature_pred_base[ind_threshold] + 1
|
||||
|
||||
if y_test[ind] == pred[ind]:
|
||||
right_base[ind_threshold] = right_base[ind_threshold] + 1
|
||||
if int(y_test[ind]) == 1:
|
||||
feature_right_base[ind_threshold] = feature_right_base[ind_threshold] + 1
|
||||
if ind_threshold == 0:
|
||||
low_threshold = -0.1
|
||||
else:
|
||||
low_threshold = threshold[ind_threshold]- 1.0/break_count
|
||||
if abs(diff) <= threshold[ind_threshold] and abs(diff) > low_threshold:
|
||||
count_base2[ind_threshold] = count_base2[ind_threshold] + 1
|
||||
if int(y_test[ind]) == 1:
|
||||
feature_count_base2[ind_threshold] = feature_count_base2[ind_threshold] +1
|
||||
if int(pred[ind]) == 1:
|
||||
feature_pred_base2[ind_threshold] = feature_pred_base2[ind_threshold] + 1
|
||||
|
||||
if y_test[ind] == pred[ind]:
|
||||
right_base2[ind_threshold] = right_base2[ind_threshold] + 1
|
||||
if int(y_test[ind]) == 1:
|
||||
feature_right_base2[ind_threshold] = feature_right_base2[ind_threshold] + 1
|
||||
|
||||
# flag for diff
|
||||
flag = 'zero'
|
||||
if diff > 0:
|
||||
flag = "+"
|
||||
elif diff < 0:
|
||||
flag = "-"
|
||||
# print("id:"+repr(x_id[ind]))
|
||||
if len(x_pred)>0:
|
||||
for j in range(len(x_pred)):
|
||||
if(helper.word_count(sentences[j])>3):
|
||||
# recording which to change
|
||||
for ind_change in range(change_break):
|
||||
if pred[ind] == 0 and x_prob[j:j+1,1][0] > ind_change*1.0/change_break:
|
||||
change_flag(change_flag_set[0],ind_threshold,ind_change,j,sentences_count)
|
||||
if pred[ind] == 0 and x_prob[j:j+1,1][0] > ind_change*1.0/change_break and flag != '+':
|
||||
change_flag(change_flag_set[1],ind_threshold,ind_change,j,sentences_count)
|
||||
if pred[ind] == 0 and x_prob[j:j+1,1][0] > ind_change*1.0/change_break and flag == '-':
|
||||
change_flag(change_flag_set[2],ind_threshold,ind_change,j,sentences_count)
|
||||
# if pred[ind] == 1 and x_prob[j:j+1,0][0] > ind_change*1.0/change_break and flag == '+':
|
||||
# change_flag(change_flag_set[3],ind_threshold,ind_change)
|
||||
|
||||
status = (y_test[ind] == pred[ind])
|
||||
data_process(change_flag_set,ind_threshold,status)
|
||||
for ind_change in range(change_break):
|
||||
if change_flag_set[0][ind_threshold][ind_change]:
|
||||
feature_count_mine1[ind_threshold][ind_change] = feature_count_mine1[ind_threshold][ind_change] + 1
|
||||
if change_flag_set[1][ind_threshold][ind_change]:
|
||||
feature_count_mine2[ind_threshold][ind_change] = feature_count_mine2[ind_threshold][ind_change] + 1
|
||||
if change_flag_set[2][ind_threshold][ind_change]:
|
||||
feature_count_mine3[ind_threshold][ind_change] = feature_count_mine3[ind_threshold][ind_change] + 1
|
||||
if change_flag_set[3][ind_threshold][ind_change]:
|
||||
feature_count_mine4[ind_threshold][ind_change] = feature_count_mine4[ind_threshold][ind_change] + 1
|
||||
|
||||
data_process2(change_flag_set,ind_threshold,status)
|
||||
|
||||
# issue.close()
|
||||
clf_descr = str(clf).split('(')[0]
|
||||
|
||||
return clf_descr, score, train_time, test_time
|
||||
##############################################
|
||||
results.append(benchmark(svm.SVC(kernel='linear',probability=True)))
|
||||
|
||||
results = [[x[i] for x in results] for i in range(4)]
|
||||
clf_names, score, training_time, test_time = results
|
||||
|
||||
info_len = len(score)
|
||||
training_time = np.array(training_time).sum() / info_len
|
||||
test_time = np.array(test_time).sum() / info_len
|
||||
score_all = np.array(score).sum()/ info_len
|
||||
print("accuracy for all: %0.3f" % score_all)
|
||||
writer_classifier.writerow(['------------------------------'])
|
||||
writer_classifier.writerow(["traning time for all:", training_time])
|
||||
writer_classifier.writerow(["test time for all:", test_time])
|
||||
writer_classifier.writerow(["accuracy for all:", score_all])
|
||||
|
||||
# write result
|
||||
writer_result.writerow(['base count for all:'])
|
||||
writer_result.writerow([n for n in count_base])
|
||||
writer_result.writerow(['base right count for all:'])
|
||||
writer_result.writerow([n for n in right_base])
|
||||
|
||||
writer_result.writerow(['base count for threshold:'])
|
||||
writer_result.writerow([n for n in count_base2])
|
||||
writer_result.writerow(['base right count for threshold:'])
|
||||
writer_result.writerow([n for n in right_base2])
|
||||
|
||||
|
||||
feature_writer_result.writerow(['base count for all:'])
|
||||
feature_writer_result.writerow([n for n in feature_count_base])
|
||||
feature_writer_result.writerow(['base right count for all:'])
|
||||
feature_writer_result.writerow([n for n in feature_right_base])
|
||||
feature_writer_result.writerow(['base pred count for all:'])
|
||||
feature_writer_result.writerow([n for n in feature_pred_base])
|
||||
|
||||
feature_writer_result.writerow(['base count for threshold:'])
|
||||
feature_writer_result.writerow([n for n in feature_count_base2])
|
||||
feature_writer_result.writerow(['base right count for threshold:'])
|
||||
feature_writer_result.writerow([n for n in feature_right_base2])
|
||||
feature_writer_result.writerow(['base pred count for threshold:'])
|
||||
feature_writer_result.writerow([n for n in feature_pred_base2])
|
||||
|
||||
writer_result.writerow(['right count of mine method 1 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
writer_result.writerow([l for l in right_mine1[i]])
|
||||
writer_result.writerow(['right count of mine method 2 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
writer_result.writerow([l for l in right_mine2[i]])
|
||||
writer_result.writerow(['right count of mine method 3 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
writer_result.writerow([l for l in right_mine3[i]])
|
||||
writer_result.writerow(['right count of mine method 4 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
writer_result.writerow([l for l in right_mine4[i]])
|
||||
|
||||
|
||||
feature_writer_result.writerow(['right count of mine method 1 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
feature_writer_result.writerow([l for l in feature_right_mine1[i]])
|
||||
feature_writer_result.writerow(['right count of mine method 2 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
feature_writer_result.writerow([l for l in feature_right_mine2[i]])
|
||||
feature_writer_result.writerow(['right count of mine method 3 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
feature_writer_result.writerow([l for l in feature_right_mine3[i]])
|
||||
feature_writer_result.writerow(['right count of mine method 4 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
feature_writer_result.writerow([l for l in feature_right_mine4[i]])
|
||||
|
||||
feature_writer_result.writerow(['change count of mine method 1 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
feature_writer_result.writerow([l for l in feature_count_mine1[i]])
|
||||
feature_writer_result.writerow(['change count of mine method 2 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
feature_writer_result.writerow([l for l in feature_count_mine2[i]])
|
||||
feature_writer_result.writerow(['change count of mine method 3 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
feature_writer_result.writerow([l for l in feature_count_mine3[i]])
|
||||
feature_writer_result.writerow(['change count of mine method 4 for each threshold:'])
|
||||
for i in range(break_count):
|
||||
feature_writer_result.writerow([l for l in feature_count_mine4[i]])
|
||||
|
||||
csv_result.close()
|
||||
csv_path.close()
|
||||
csv_classifier.close()
|
||||
|
||||
|
||||
# classifier_project_by_id('6')
|
||||
|
||||
# break_id = 961
|
||||
# projects = dao.get_project()
|
||||
# flag_break = True
|
||||
# csv_project = file('project_id.csv', 'wb')
|
||||
# writer_project = csv.writer(csv_project)
|
||||
# for project in projects:
|
||||
# print('do classifier for project:'+ repr(project[0]))
|
||||
# if project[0] == break_id:
|
||||
# flag_break = True
|
||||
#
|
||||
# if project[1] > 500 and flag_break:
|
||||
# try:
|
||||
# classifier_project_by_id(project[0])
|
||||
# writer_project.writerow(project)
|
||||
# except:
|
||||
# f=open("log.txt",'a')
|
||||
# f.writelines("project:\t"+repr(project[0])+'\n')
|
||||
# f.flush()
|
||||
# f.close()
|
||||
#
|
||||
# csv_project.close()
|
||||
# projects.close()
|
||||
|
||||
f_proj_id = file('proj_id.csv', 'r')
|
||||
reader = csv.reader(f_proj_id)
|
||||
for line in reader:
|
||||
classifier_project_by_id(line[0])
|
||||
|
||||
print(line[0])
|
||||
dao.close()
|
||||
Loading…
Reference in New Issue