Compare commits
2 Commits
feature_yu
...
master
| Author | SHA1 | Date |
|---|---|---|
|
|
9caa669897 | |
|
|
e9de000618 |
|
|
@ -61,10 +61,10 @@
|
|||
</provider>
|
||||
</entry>
|
||||
</file>
|
||||
<file leaf-file-name="helper.py" pinned="false" current-in-tab="true">
|
||||
<file leaf-file-name="helper.py" pinned="false" current-in-tab="false">
|
||||
<entry file="file://$PROJECT_DIR$/helper.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="1.7916666">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="120" column="13" selection-start-line="120" selection-start-column="13" selection-end-line="120" selection-end-column="13" />
|
||||
<folding />
|
||||
</state>
|
||||
|
|
@ -98,7 +98,7 @@
|
|||
<file leaf-file-name="sentence_classifier2.py" pinned="false" current-in-tab="false">
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier2.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="-49.121952">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="437" column="11" selection-start-line="437" selection-start-column="11" selection-end-line="437" selection-end-column="11" />
|
||||
<folding />
|
||||
</state>
|
||||
|
|
@ -127,27 +127,10 @@
|
|||
</provider>
|
||||
</entry>
|
||||
</file>
|
||||
<file leaf-file-name="classifier_for_yu.py" pinned="false" current-in-tab="false">
|
||||
<entry file="file://$PROJECT_DIR$/classifier_for_yu.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="87" column="43" selection-start-line="87" selection-start-column="43" selection-end-line="87" selection-end-column="43" />
|
||||
<folding>
|
||||
<marker date="1453738619885" expanded="true" signature="2307:2338" placeholder="select title..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2307:2344" placeholder="select title..numpy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2735" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2736" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2785" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2786" placeholder="update phpmy..." />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
</file>
|
||||
<file leaf-file-name="dao.py" pinned="false" current-in-tab="false">
|
||||
<entry file="file://$PROJECT_DIR$/dao.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="24.066668">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="4" column="14" selection-start-line="4" selection-start-column="7" selection-end-line="4" selection-end-column="14" />
|
||||
<folding>
|
||||
<marker date="1453964958671" expanded="true" signature="252:294" placeholder="select title..issues..." />
|
||||
|
|
@ -169,6 +152,16 @@
|
|||
</provider>
|
||||
</entry>
|
||||
</file>
|
||||
<file leaf-file-name="read me.txt" pinned="false" current-in-tab="true">
|
||||
<entry file="file://$PROJECT_DIR$/read me.txt">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.084916204">
|
||||
<caret line="4" column="73" selection-start-line="4" selection-start-column="73" selection-end-line="4" selection-end-column="73" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
</file>
|
||||
</leaf>
|
||||
</component>
|
||||
<component name="FileTemplateManagerImpl">
|
||||
|
|
@ -196,6 +189,7 @@
|
|||
<option value="$PROJECT_DIR$/kmeans.py" />
|
||||
<option value="$PROJECT_DIR$/dao.py" />
|
||||
<option value="$PROJECT_DIR$/lda_example.py" />
|
||||
<option value="$PROJECT_DIR$/read me.txt" />
|
||||
</list>
|
||||
</option>
|
||||
</component>
|
||||
|
|
@ -235,7 +229,7 @@
|
|||
<sortByType />
|
||||
</navigator>
|
||||
<panes>
|
||||
<pane id="Scratches" />
|
||||
<pane id="Scope" />
|
||||
<pane id="ProjectPane">
|
||||
<subPane>
|
||||
<PATH>
|
||||
|
|
@ -256,7 +250,7 @@
|
|||
</PATH>
|
||||
</subPane>
|
||||
</pane>
|
||||
<pane id="Scope" />
|
||||
<pane id="Scratches" />
|
||||
</panes>
|
||||
</component>
|
||||
<component name="PropertiesComponent">
|
||||
|
|
@ -577,28 +571,28 @@
|
|||
<servers />
|
||||
</component>
|
||||
<component name="ToolWindowManager">
|
||||
<frame x="-8" y="-8" width="1936" height="1056" extended-state="6" />
|
||||
<frame x="-8" y="-8" width="1936" height="1056" extended-state="0" />
|
||||
<editor active="true" />
|
||||
<layout>
|
||||
<window_info id="Project" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" weight="0.15991472" sideWeight="0.5" order="0" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Project" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" weight="0.16151386" sideWeight="0.5" order="0" side_tool="false" content_ui="combo" />
|
||||
<window_info id="TODO" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="6" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Event Log" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.32792208" sideWeight="0.52558637" order="7" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="Application Servers" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="7" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Database" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Python Console" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="7" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Version Control" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="7" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Run" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" weight="0.16341992" sideWeight="0.47441363" order="2" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Structure" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.25" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Terminal" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.32900432" sideWeight="0.48507464" order="7" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Favorites" active="false" anchor="left" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="2" side_tool="true" content_ui="tabs" />
|
||||
<window_info id="Debug" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.30844155" sideWeight="0.7677039" order="3" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Cvs" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.25" sideWeight="0.5" order="4" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Message" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="0" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Commander" active="false" anchor="right" auto_hide="false" internal_type="SLIDING" type="SLIDING" visible="false" weight="0.4" sideWeight="0.5" order="0" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Inspection" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.4" sideWeight="0.5" order="5" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Run" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="true" weight="0.16341992" sideWeight="0.47441363" order="2" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Hierarchy" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.25" sideWeight="0.5" order="2" side_tool="false" content_ui="combo" />
|
||||
<window_info id="Find" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Ant Build" active="false" anchor="right" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.25" sideWeight="0.5" order="1" side_tool="false" content_ui="tabs" />
|
||||
<window_info id="Debug" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.30844155" sideWeight="0.7677039" order="3" side_tool="false" content_ui="tabs" />
|
||||
</layout>
|
||||
<layout-to-restore>
|
||||
<window_info id="TODO" active="false" anchor="bottom" auto_hide="false" internal_type="DOCKED" type="DOCKED" visible="false" weight="0.33" sideWeight="0.5" order="6" side_tool="false" content_ui="tabs" />
|
||||
|
|
@ -666,11 +660,43 @@
|
|||
</watches-manager>
|
||||
</component>
|
||||
<component name="editorHistoryManager">
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="415" column="0" selection-start-line="415" selection-start-column="0" selection-end-line="415" selection-end-column="0" />
|
||||
<folding>
|
||||
<element signature="e#19#30#0" expanded="true" />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/dao.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="4" column="14" selection-start-line="4" selection-start-column="7" selection-end-line="4" selection-end-column="14" />
|
||||
<folding>
|
||||
<marker date="1453964958671" expanded="true" signature="252:294" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="470:608" placeholder="select issue..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="708:822" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="931:1001" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="931:1060" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="1156:1262" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="2082:2209" placeholder="select proje..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="2520:2635" placeholder="select title..issues_for_yu..." />
|
||||
<marker date="1453964958671" expanded="true" signature="2863:2937" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3193:3259" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3193:3297" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3193:3298" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3325:3400" placeholder="update numpy..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3325:3417" placeholder="update piwik..." />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/get_analysis_result.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="187" column="36" selection-start-line="187" selection-start-column="36" selection-end-line="187" selection-end-column="36" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
|
|
@ -698,10 +724,6 @@
|
|||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="40" column="13" selection-start-line="40" selection-start-column="13" selection-end-line="40" selection-end-column="13" />
|
||||
<folding>
|
||||
<marker date="1453684270809" expanded="true" signature="-1:-1" placeholder="select title..." />
|
||||
<marker date="1453684270809" expanded="true" signature="-1:-1" placeholder="select title..." />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
|
|
@ -739,21 +761,6 @@
|
|||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/classifier_for_yu.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="0" column="0" selection-start-line="0" selection-start-column="0" selection-end-line="0" selection-end-column="0" />
|
||||
<folding>
|
||||
<marker date="1453738619885" expanded="true" signature="2307:2338" placeholder="select title..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2307:2344" placeholder="select title..numpy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2735" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2736" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2785" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2786" placeholder="update phpmy..." />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/dao.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
|
|
@ -781,7 +788,6 @@
|
|||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="185" column="27" selection-start-line="185" selection-start-column="27" selection-end-line="185" selection-end-column="27" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
|
|
@ -842,22 +848,6 @@
|
|||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="0" column="0" selection-start-line="0" selection-start-column="0" selection-end-line="0" selection-end-column="0" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/classifier_for_yu.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="59" column="34" selection-start-line="59" selection-start-column="34" selection-end-line="59" selection-end-column="34" />
|
||||
<folding>
|
||||
<marker date="1453738619885" expanded="true" signature="2307:2338" placeholder="select title..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2307:2344" placeholder="select title..numpy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2735" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2736" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2785" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2786" placeholder="update phpmy..." />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
|
|
@ -921,7 +911,6 @@
|
|||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="0" column="0" selection-start-line="0" selection-start-column="0" selection-end-line="0" selection-end-column="0" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
|
|
@ -977,7 +966,6 @@
|
|||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="159" column="24" selection-start-line="159" selection-start-column="24" selection-end-line="159" selection-end-column="24" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
|
|
@ -1047,7 +1035,6 @@
|
|||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="0" column="0" selection-start-line="0" selection-start-column="0" selection-end-line="0" selection-end-column="0" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
|
|
@ -1069,10 +1056,6 @@
|
|||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="40" column="13" selection-start-line="40" selection-start-column="13" selection-end-line="40" selection-end-column="13" />
|
||||
<folding>
|
||||
<marker date="1453684270809" expanded="true" signature="-1:-1" placeholder="select title..." />
|
||||
<marker date="1453684270809" expanded="true" signature="-1:-1" placeholder="select title..." />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
|
|
@ -1080,7 +1063,6 @@
|
|||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="187" column="36" selection-start-line="187" selection-start-column="36" selection-end-line="187" selection-end-column="36" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
|
|
@ -1122,47 +1104,9 @@
|
|||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/classifier_for_yu.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="87" column="43" selection-start-line="87" selection-start-column="43" selection-end-line="87" selection-end-column="43" />
|
||||
<folding>
|
||||
<marker date="1453738619885" expanded="true" signature="2307:2338" placeholder="select title..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2307:2344" placeholder="select title..numpy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2735" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2736" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2785" placeholder="update phpmy..." />
|
||||
<marker date="1453738619885" expanded="true" signature="2702:2786" placeholder="update phpmy..." />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/dao.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="24.066668">
|
||||
<caret line="4" column="14" selection-start-line="4" selection-start-column="7" selection-end-line="4" selection-end-column="14" />
|
||||
<folding>
|
||||
<marker date="1453964958671" expanded="true" signature="252:294" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="470:608" placeholder="select issue..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="708:822" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="931:1001" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="931:1060" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="1156:1262" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="2082:2209" placeholder="select proje..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="2520:2635" placeholder="select title..issues_for_yu..." />
|
||||
<marker date="1453964958671" expanded="true" signature="2863:2937" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3193:3259" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3193:3297" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3193:3298" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3325:3400" placeholder="update numpy..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3325:3417" placeholder="update piwik..." />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/sentence_classifier2.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="-49.121952">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="437" column="11" selection-start-line="437" selection-start-column="11" selection-end-line="437" selection-end-column="11" />
|
||||
<folding />
|
||||
</state>
|
||||
|
|
@ -1190,11 +1134,42 @@
|
|||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/helper.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="1.7916666">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="120" column="13" selection-start-line="120" selection-start-column="13" selection-end-line="120" selection-end-column="13" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/dao.py">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.0">
|
||||
<caret line="4" column="14" selection-start-line="4" selection-start-column="7" selection-end-line="4" selection-end-column="14" />
|
||||
<folding>
|
||||
<marker date="1453964958671" expanded="true" signature="252:294" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="470:608" placeholder="select issue..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="708:822" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="931:1001" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="931:1060" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="1156:1262" placeholder="select title..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="2082:2209" placeholder="select proje..issues..." />
|
||||
<marker date="1453964958671" expanded="true" signature="2520:2635" placeholder="select title..issues_for_yu..." />
|
||||
<marker date="1453964958671" expanded="true" signature="2863:2937" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3193:3259" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3193:3297" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3193:3298" placeholder="update issue..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3325:3400" placeholder="update numpy..." />
|
||||
<marker date="1453964958671" expanded="true" signature="3325:3417" placeholder="update piwik..." />
|
||||
</folding>
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
<entry file="file://$PROJECT_DIR$/read me.txt">
|
||||
<provider selected="true" editor-type-id="text-editor">
|
||||
<state vertical-scroll-proportion="0.084916204">
|
||||
<caret line="4" column="73" selection-start-line="4" selection-start-column="73" selection-end-line="4" selection-end-column="73" />
|
||||
<folding />
|
||||
</state>
|
||||
</provider>
|
||||
</entry>
|
||||
</component>
|
||||
</project>
|
||||
100
analysis.py
100
analysis.py
|
|
@ -38,19 +38,19 @@ def get_analysis(proj_id,method):
|
|||
easy_right = easy_right+1
|
||||
return hard_right,hard_count,easy_right,len(different)-hard_count,acc
|
||||
|
||||
method = 'svm'
|
||||
f_proj_id = file('proj_id.csv', 'r')
|
||||
reader = csv.reader(f_proj_id)
|
||||
final_path = 'final2/'
|
||||
helper.mkdir(final_path)
|
||||
f_hard_tongji = file(final_path + 'hard_count_'+method+'.csv','w')
|
||||
writer_final = csv.writer(f_hard_tongji)
|
||||
writer_final.writerow(['proj_id','hard_right','hard_count','easy_right','easy_count','acc'])
|
||||
for line in reader:
|
||||
temp = get_analysis(line[0],method)
|
||||
data = (line[0],temp[0],temp[1],temp[2],temp[3],temp[4])
|
||||
writer_final.writerow(data)
|
||||
f_hard_tongji.close()
|
||||
# method = 'svm'
|
||||
# f_proj_id = file('proj_id.csv', 'r')
|
||||
# reader = csv.reader(f_proj_id)
|
||||
# final_path = 'final2/'
|
||||
# helper.mkdir(final_path)
|
||||
# f_hard_tongji = file(final_path + 'hard_count_'+method+'.csv','w')
|
||||
# writer_final = csv.writer(f_hard_tongji)
|
||||
# writer_final.writerow(['proj_id','hard_right','hard_count','easy_right','easy_count','acc'])
|
||||
# for line in reader:
|
||||
# temp = get_analysis(line[0],method)
|
||||
# data = (line[0],temp[0],temp[1],temp[2],temp[3],temp[4])
|
||||
# writer_final.writerow(data)
|
||||
# f_hard_tongji.close()
|
||||
|
||||
# get result of whose id = n
|
||||
def get_result_of_n(n):
|
||||
|
|
@ -72,7 +72,7 @@ def get_result_of_n(n):
|
|||
|
||||
# get title and description for selected issues and write in test.csv
|
||||
conn = pymysql.connect(host='127.0.0.1', port=3306, user='root', passwd='123456', db='zlb_github')
|
||||
def get_title_description(issue_id,y_test,pred):
|
||||
def get_title_description(issue_id,y_test,pred,path):
|
||||
print("path:" + path + 'csv_hard.csv')
|
||||
csvfile = file(path + 'csv_hard.csv', 'wb')
|
||||
writer = csv.writer(csvfile)
|
||||
|
|
@ -80,8 +80,9 @@ def get_title_description(issue_id,y_test,pred):
|
|||
|
||||
cur = conn.cursor()
|
||||
for i in range(len(issue_id)):
|
||||
sql = "select title,body from "\
|
||||
# sql = "select title,body from "\
|
||||
# +table+" where id = " + str(issue_id[i])
|
||||
sql = 'select title, body from issues where id = '+repr(issue_id[i])
|
||||
cur.execute(sql)
|
||||
r = cur.fetchone()
|
||||
if r:
|
||||
|
|
@ -107,7 +108,8 @@ def get_result_by_variance(n):
|
|||
|
||||
# get result by set threshold of different of max and second
|
||||
def get_result_by_different(n, count=0, count_diff=0, count_sth=0, count_all=0):
|
||||
path = ''
|
||||
proj_id = "6013"
|
||||
path = 'result/'+proj_id+"/svm/data/"
|
||||
print("-"*100)
|
||||
print("get result by diff:"+repr(n))
|
||||
x_id_set = []
|
||||
|
|
@ -116,30 +118,33 @@ def get_result_by_different(n, count=0, count_diff=0, count_sth=0, count_all=0):
|
|||
for i in range(1,10,1):
|
||||
y_test = np.load(path + 'y_test_' + repr(i) + ".npy")
|
||||
pred = np.load(path + 'pred_' + repr(i) + ".npy")
|
||||
variance = np.load(path + 'variance' + repr(i) + ".npy")
|
||||
different = np.load(path + 'different' + repr(i) + ".npy")
|
||||
probability = np.load(path + 'probability' + repr(i) + ".npy")
|
||||
# variance = np.load(path + 'variance' + repr(i) + ".npy")
|
||||
# different = np.load(path + 'different' + repr(i) + ".npy")
|
||||
probability = np.load(path + 'probability_' + repr(i) + ".npy")
|
||||
x_id = np.load(path + 'x_id_' + repr(i) + ".npy")
|
||||
for j in range(len(x_id)):
|
||||
count_all += 1
|
||||
if pred[j:j+1] == 1:
|
||||
count_sth += 1
|
||||
if different[j:j+1]>=n:
|
||||
diff = abs(probability[j:j+1][0][0] - probability[j:j+1][0][1])
|
||||
if diff <= n:
|
||||
count += 1
|
||||
# print("x_id:"+repr(x_id[j:j+1].tolist())+"\ty_test:"+repr(y_test[j:j+1].tolist())+"\tpred:"+repr(pred[j:j+1].tolist())+"\tvariance:"+repr(variance[j:j+1].tolist())+"\tdifferent:"+repr(different[j:j+1].tolist()))
|
||||
# print(probability[j:j+1])
|
||||
if y_test[j:j+1] != pred[j:j+1]:
|
||||
print("x_id:"+repr(x_id[j:j+1].tolist())+"\ty_test:"+repr(y_test[j:j+1].tolist())+"\tpred:"+repr(pred[j:j+1].tolist())+"\tvariance:"+repr(variance[j:j+1].tolist())+"\tdifferent:"+repr(different[j:j+1].tolist()))
|
||||
print(probability[j:j+1])
|
||||
count_diff += 1
|
||||
x_id_set.append(x_id[j:j+1].tolist()[0])
|
||||
y_test_set.append(y_test[j:j+1].tolist()[0])
|
||||
pred_set.append(pred[j:j+1].tolist()[0])
|
||||
# if y_test[j:j+1] != pred[j:j+1]:
|
||||
# print("x_id:"+repr(x_id[j:j+1].tolist())+"\ty_test:"+repr(y_test[j:j+1].tolist())+"\tpred:"+repr(pred[j:j+1].tolist())+"\tvariance:"+repr(variance[j:j+1].tolist())+"\tdifferent:"+repr(different[j:j+1].tolist()))
|
||||
print(probability[j:j+1])
|
||||
count_diff += 1
|
||||
x_id_set.append(x_id[j:j+1].tolist()[0])
|
||||
y_test_set.append(y_test[j:j+1].tolist()[0])
|
||||
pred_set.append(pred[j:j+1].tolist()[0])
|
||||
print("different > n data count:"+repr(count))
|
||||
print("different > n and wrong pred data count:"+repr(count_diff))
|
||||
print("count sth:"+repr(count_sth))
|
||||
print("all issue count:"+repr(count_all))
|
||||
get_title_description(x_id_set,y_test_set,pred_set)
|
||||
get_title_description(x_id_set,y_test_set,pred_set,path)
|
||||
|
||||
# get_result_by_different(0.1)
|
||||
|
||||
# get result by set threshold of different of max and second
|
||||
def get_result_by_little_different(n, count=0, count_diff=0, count_sth=0, count_all=0):
|
||||
|
|
@ -173,46 +178,55 @@ def get_result_by_little_different(n, count=0, count_diff=0, count_sth=0, count_
|
|||
print("all issue count:"+repr(count_all))
|
||||
get_title_description(x_id_set,y_test_set,pred_set)
|
||||
|
||||
def get_all_pred_analysis(count_all=0):
|
||||
# ####################################################################
|
||||
def get_all_pred_analysis(count_all=0,proj_id = "0"):
|
||||
|
||||
path = 'result3/'+proj_id+"/svm/data/"
|
||||
x_id_set = []
|
||||
y_test_set = []
|
||||
pred_set = []
|
||||
variance_set = []
|
||||
# variance_set = []
|
||||
different_set = []
|
||||
probability_set = []
|
||||
for i in range(1,10,1):
|
||||
y_test = np.load(path + 'y_test_' + repr(i) + ".npy")
|
||||
pred = np.load(path + 'pred_' + repr(i) + ".npy")
|
||||
variance = np.load(path + 'variance' + repr(i) + ".npy")
|
||||
different = np.load(path + 'different' + repr(i) + ".npy")
|
||||
probability = np.load(path + 'probability' + repr(i) + ".npy")
|
||||
# variance = np.load(path + 'variance' + repr(i) + ".npy")
|
||||
# different = np.load(path + 'different' + repr(i) + ".npy")
|
||||
probability = np.load(path + 'probability_' + repr(i) + ".npy")
|
||||
x_id = np.load(path + 'x_id_' + repr(i) + ".npy")
|
||||
for j in range(len(x_id)):
|
||||
count_all += 1
|
||||
x_id_set.append(x_id[j:j+1].tolist()[0])
|
||||
y_test_set.append(y_test[j:j+1].tolist()[0])
|
||||
pred_set.append(pred[j:j+1].tolist()[0])
|
||||
variance_set.append(variance[j:j+1].tolist()[0])
|
||||
different_set.append(different[j:j+1].tolist()[0])
|
||||
# variance_set.append(variance[j:j+1].tolist()[0])
|
||||
different_set.append(abs(probability[j:j+1][0][0] - probability[j:j+1][0][1]))
|
||||
probability_set.append(probability[j:j+1].tolist()[0])
|
||||
print("all issue count:"+repr(count_all))
|
||||
print("path:" + path + 'csv_analysis.csv')
|
||||
csvfile = file(path + 'csv_analysis.csv', 'wb')
|
||||
writer = csv.writer(csvfile)
|
||||
# writer.writerow(['id', 'y_test', 'pred', 'variance', 'different', 'probability', 'title', 'body'])
|
||||
writer.writerow(['id', 'y_test', 'pred', 'variance', 'different', 'probability'])
|
||||
writer.writerow(['id', 'y_test', 'pred', 'different', 'probability'])
|
||||
|
||||
cur = conn.cursor()
|
||||
for i in range(len(x_id_set)):
|
||||
sql = "select title,body from "\
|
||||
# +table+" where id = " + str(x_id_set[i])
|
||||
cur.execute(sql)
|
||||
r = cur.fetchone()
|
||||
if r:
|
||||
data = (x_id_set[i], y_test_set[i], pred_set[i], variance_set[i], different_set[i], probability_set[i])
|
||||
writer.writerow(data)
|
||||
# sql = "select title,body from "\
|
||||
# # +table+" where id = " + str(x_id_set[i])
|
||||
# cur.execute(sql)
|
||||
# r = cur.fetchone()
|
||||
# if r:
|
||||
data = (x_id_set[i], y_test_set[i], pred_set[i], different_set[i], probability_set[i])
|
||||
writer.writerow(data)
|
||||
csvfile.close()
|
||||
|
||||
# proj_id = "11450"
|
||||
# proj_id = "6013"
|
||||
proj_id = "24444"
|
||||
|
||||
get_all_pred_analysis(0,proj_id)
|
||||
|
||||
# def get_pred_rate(break = 10):
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1,8 +1,8 @@
|
|||
classifier.py:对数据进行预处理,包括stemming,去除数字,特殊字符,TFIDF计算
|
||||
sentence_classifier.py:分类程序,包括直接使用svm,对hard部分的特殊处理,结果统计与持久化
|
||||
dao:数据库访问方法
|
||||
dao:数据库访问方法,以及数据库连接设置
|
||||
helper:一些复杂方法的实现(包括特殊符号/数字筛选,代码识别,文件数据读取等)
|
||||
get_analysis_result/get_feature_analysis_result:对分类结果的分析
|
||||
get_analysis_result/get_feature_analysis_result:对分类结果的分析(包括准确率提升,修改的缺陷数量等)
|
||||
lda_example:lda主题模型+kmeans
|
||||
read_lda:读取lda模型
|
||||
kmeans:用tfidf做kmeans
|
||||
|
|
|
|||
|
|
@ -0,0 +1,63 @@
|
|||
/*
|
||||
Navicat MySQL Data Transfer
|
||||
|
||||
Source Server : qiangge
|
||||
Source Server Version : 50537
|
||||
Source Host : localhost:3306
|
||||
Source Database : zlb_github
|
||||
|
||||
Target Server Type : MYSQL
|
||||
Target Server Version : 50537
|
||||
File Encoding : 65001
|
||||
|
||||
Date: 2016-01-31 14:20:28
|
||||
*/
|
||||
|
||||
SET FOREIGN_KEY_CHECKS=0;
|
||||
|
||||
-- ----------------------------
|
||||
-- Table structure for issues
|
||||
-- ----------------------------
|
||||
DROP TABLE IF EXISTS `issues`;
|
||||
CREATE TABLE `issues` (
|
||||
`body` text,
|
||||
`events_url` varchar(200) DEFAULT NULL,
|
||||
`locked` int(5) DEFAULT NULL,
|
||||
`title` text,
|
||||
`url` varchar(200) DEFAULT NULL,
|
||||
`labels_url` varchar(200) DEFAULT NULL,
|
||||
`created_at` varchar(200) DEFAULT NULL,
|
||||
`comments_url` varchar(200) DEFAULT NULL,
|
||||
`html_url` varchar(200) DEFAULT NULL,
|
||||
`comments` int(11) DEFAULT NULL,
|
||||
`number` int(11) DEFAULT NULL,
|
||||
`updated_at` varchar(100) DEFAULT NULL,
|
||||
`state` varchar(100) DEFAULT NULL,
|
||||
`closed_at` varchar(100) DEFAULT NULL,
|
||||
`project_id` int(11) DEFAULT NULL,
|
||||
`id` int(100) NOT NULL AUTO_INCREMENT,
|
||||
`user.following_url` varchar(200) DEFAULT NULL,
|
||||
`user.gists_url` varchar(200) DEFAULT NULL,
|
||||
`user.organizations_url` varchar(200) DEFAULT NULL,
|
||||
`user.url` varchar(200) DEFAULT NULL,
|
||||
`user.events_url` varchar(200) DEFAULT NULL,
|
||||
`user.html_url` varchar(200) DEFAULT NULL,
|
||||
`user.subscriptions_url` varchar(200) DEFAULT NULL,
|
||||
`user.avatar_url` varchar(200) DEFAULT NULL,
|
||||
`user.repos_url` varchar(200) DEFAULT NULL,
|
||||
`user.received_events_url` varchar(200) DEFAULT NULL,
|
||||
`user.gravatar_id` varchar(200) DEFAULT NULL,
|
||||
`user.starred_url` varchar(200) DEFAULT NULL,
|
||||
`user.site_admin` int(11) DEFAULT NULL,
|
||||
`user.login` varchar(100) DEFAULT NULL,
|
||||
`user.type` varchar(50) DEFAULT NULL,
|
||||
`user.id` varchar(100) DEFAULT NULL,
|
||||
`user.followers_url` varchar(200) DEFAULT NULL,
|
||||
`issue_type` varchar(255) DEFAULT NULL,
|
||||
`kmeans` int(11) DEFAULT NULL,
|
||||
PRIMARY KEY (`id`),
|
||||
KEY `proj_id` (`project_id`),
|
||||
KEY `issue_type` (`issue_type`(250)),
|
||||
KEY `number` (`number`),
|
||||
KEY `kmeans_index` (`kmeans`)
|
||||
) ENGINE=MyISAM AUTO_INCREMENT=123846131 DEFAULT CHARSET=utf8mb4;
|
||||
|
|
@ -0,0 +1,31 @@
|
|||
/*
|
||||
Navicat MySQL Data Transfer
|
||||
|
||||
Source Server : qiangge
|
||||
Source Server Version : 50537
|
||||
Source Host : localhost:3306
|
||||
Source Database : zlb_github
|
||||
|
||||
Target Server Type : MYSQL
|
||||
Target Server Version : 50537
|
||||
File Encoding : 65001
|
||||
|
||||
Date: 2016-01-31 14:20:50
|
||||
*/
|
||||
|
||||
SET FOREIGN_KEY_CHECKS=0;
|
||||
|
||||
-- ----------------------------
|
||||
-- Table structure for issues_labels
|
||||
-- ----------------------------
|
||||
DROP TABLE IF EXISTS `issues_labels`;
|
||||
CREATE TABLE `issues_labels` (
|
||||
`url` varchar(200) DEFAULT NULL,
|
||||
`color` varchar(100) DEFAULT NULL,
|
||||
`name` varchar(100) NOT NULL,
|
||||
`project_id` bigint(50) NOT NULL,
|
||||
`issues_number` bigint(50) NOT NULL,
|
||||
`issues_id` int(50) DEFAULT '0',
|
||||
PRIMARY KEY (`project_id`,`name`,`issues_number`),
|
||||
KEY `label` (`name`)
|
||||
) ENGINE=MyISAM DEFAULT CHARSET=utf8mb4;
|
||||
|
|
@ -0,0 +1,32 @@
|
|||
/*
|
||||
Navicat MySQL Data Transfer
|
||||
|
||||
Source Server : qiangge
|
||||
Source Server Version : 50537
|
||||
Source Host : localhost:3306
|
||||
Source Database : zlb_github
|
||||
|
||||
Target Server Type : MYSQL
|
||||
Target Server Version : 50537
|
||||
File Encoding : 65001
|
||||
|
||||
Date: 2016-01-31 14:21:33
|
||||
*/
|
||||
|
||||
SET FOREIGN_KEY_CHECKS=0;
|
||||
|
||||
-- ----------------------------
|
||||
-- Table structure for numpy_sentences
|
||||
-- ----------------------------
|
||||
DROP TABLE IF EXISTS `numpy_sentences`;
|
||||
CREATE TABLE `numpy_sentences` (
|
||||
`project_id` int(11) DEFAULT NULL,
|
||||
`number` int(11) DEFAULT NULL,
|
||||
`text` text CHARACTER SET utf8mb4,
|
||||
`issue_type` varchar(255) CHARACTER SET utf8mb4 DEFAULT NULL,
|
||||
`kmeans` int(11) DEFAULT NULL,
|
||||
`id` int(11) DEFAULT NULL,
|
||||
`y_test` int(11) DEFAULT NULL,
|
||||
`pred` int(11) DEFAULT NULL,
|
||||
`diff` double DEFAULT NULL
|
||||
) ENGINE=MyISAM DEFAULT CHARSET=latin1;
|
||||
62
test.py
62
test.py
|
|
@ -1,14 +1,14 @@
|
|||
# -*- coding: utf-8 -*
|
||||
import re
|
||||
|
||||
import pymysql
|
||||
import dao
|
||||
import helper
|
||||
import nltk
|
||||
a = dao.get_info_by_id(14044309).fetchone()
|
||||
str = a[0]+".\n"+a[1]
|
||||
# str = '```asdf```asdf'
|
||||
# print(str)
|
||||
print('='*80)
|
||||
# a = dao.get_info_by_id(14044309).fetchone()
|
||||
# str = a[0]+".\n"+a[1]
|
||||
# # str = '```asdf```asdf'
|
||||
# # print(str)
|
||||
# print('='*80)
|
||||
# # p = re.compile(r'`{3,}.*?`{3,}')
|
||||
# # print(p.findall(str))
|
||||
# # print p.sub(r'',str)
|
||||
|
|
@ -21,21 +21,41 @@ print('='*80)
|
|||
# s = str
|
||||
# # s = '1\n```\n2\n```'
|
||||
# s = '1\n```\n2\n```\n3```\n4```\n'
|
||||
print
|
||||
# print
|
||||
#
|
||||
# a= '2'
|
||||
# print('start')
|
||||
# print(repr(helper.filter_str(a)))
|
||||
# print('end')
|
||||
# a = 'DHE-RSA-AES-SHA\r\n\r\nHowever, I would like to avoid using SHA, and when I include the following directive in the SSLCipherSuite :!SHA the Android-ownCloud stops working.\r\n\r\nThis is a paid app and the new ciphers have been implemented in the Windows '
|
||||
# tokenizer = nltk.data.load('tokenizers/punkt/english.pickle')
|
||||
# sentences = tokenizer.tokenize(a)
|
||||
# print(sentences)
|
||||
|
||||
a= '2'
|
||||
print('start')
|
||||
print(repr(helper.filter_str(a)))
|
||||
print('end')
|
||||
a = 'DHE-RSA-AES-SHA\r\n\r\nHowever, I would like to avoid using SHA, and when I include the following directive in the SSLCipherSuite :!SHA the Android-ownCloud stops working.\r\n\r\nThis is a paid app and the new ciphers have been implemented in the Windows '
|
||||
tokenizer = nltk.data.load('tokenizers/punkt/english.pickle')
|
||||
sentences = tokenizer.tokenize(a)
|
||||
print(sentences)
|
||||
#
|
||||
# import re
|
||||
# temp = "1/ 2/3/ 4Q:5.6.7 !!\r\n?? 8\r\n 。?9.1-0。11 ,"
|
||||
# temp = temp.decode("utf8")
|
||||
# string = re.sub("[\s+\/_$%^*\-(+\"\']+|[::+——!,。?、~@#¥%……&*()]+".decode("utf8"), " ".decode("utf8"),temp)
|
||||
# # string = ' '.join(string.split())
|
||||
# print string
|
||||
|
||||
conn = pymysql.connect(host='127.0.0.1', port=3306, user='root', passwd='123456', db='qqqq')
|
||||
|
||||
def get_data():
|
||||
cur = conn.cursor()
|
||||
sql = "select p_name,id from apache"
|
||||
cur.execute(sql)
|
||||
return cur
|
||||
|
||||
apaches = get_data()
|
||||
for project in apaches.fetchall():
|
||||
id = project[1]
|
||||
name = project[0]
|
||||
temp_name = name.split('-')[0]
|
||||
cur = conn.cursor()
|
||||
up_sql = "update apache set deal = " + repr(temp_name) + " where id = " + repr(id)
|
||||
cur.execute(up_sql)
|
||||
conn.commit()
|
||||
# print temp.split('-')[0]
|
||||
|
||||
import re
|
||||
temp = "1/ 2/3/ 4Q:5.6.7 !!\r\n?? 8\r\n 。?9.1-0。11 ,"
|
||||
temp = temp.decode("utf8")
|
||||
string = re.sub("[\s+\/_$%^*\-(+\"\']+|[::+——!,。?、~@#¥%……&*()]+".decode("utf8"), " ".decode("utf8"),temp)
|
||||
string = ' '.join(string.split())
|
||||
print string
|
||||
|
|
|
|||
|
|
@ -0,0 +1,75 @@
|
|||
__author__ = 'qiangge'
|
||||
|
||||
from sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer
|
||||
from nltk.tokenize import TweetTokenizer
|
||||
import cPickle as pickle
|
||||
import helper
|
||||
import pymysql
|
||||
import dao
|
||||
import logging
|
||||
from gensim.models import word2vec
|
||||
logging.basicConfig(format='%(asctime)s : %(levelname)s : %(message)s', level=logging.INFO)
|
||||
|
||||
# path_oringan = 'result2'
|
||||
# preprocess of data
|
||||
def data_preprocess(project_name, methold_name = ''):
|
||||
|
||||
# project_name = "owncloud"
|
||||
|
||||
path = 'result_wordembedding/'+project_name+'/'
|
||||
helper.mkdir(path)
|
||||
# if len(methold_name):
|
||||
# data_path = path + methold_name + '/data/'
|
||||
# else:
|
||||
# data_path = path + 'data/'
|
||||
# helper.mkdir(data_path)
|
||||
|
||||
|
||||
print('=' * 80)
|
||||
print("get data: ")
|
||||
cur = dao.get_data(project_name)
|
||||
fetchall = cur.fetchall()
|
||||
|
||||
train_data = []
|
||||
train_target = []
|
||||
x_id = []
|
||||
for r in fetchall:
|
||||
str = helper.filter_str(r[1]+".\n"+r[2])
|
||||
# str = helper.filter_code(str)
|
||||
train_data.append(str)
|
||||
train_target.append(r[0])
|
||||
x_id.append(r[3])
|
||||
print("data length is : ", len(train_target))
|
||||
|
||||
cur.close()
|
||||
|
||||
print('_' * 80)
|
||||
print("processing data: word embedding")
|
||||
|
||||
# categories = ['bug','enhancement','feature','documentation','question','others']
|
||||
categories = ['bug','enhancement']
|
||||
y = [None]*len(train_target)
|
||||
for i in range(len(train_target)):
|
||||
y[i] = (categories.index(train_target[i]))
|
||||
# TF-IDF
|
||||
# vectorizer = TfidfVectorizer(sublinear_tf=True, max_df=0.5,
|
||||
# stop_words='english', tokenizer=helper.tokenize_help)
|
||||
# X = vectorizer.fit_transform(train_data)
|
||||
|
||||
sentences = []
|
||||
for text in train_data:
|
||||
sentences.append(helper.tokenize_help(text))
|
||||
model = word2vec.Word2Vec(sentences,size=200,sg=1)
|
||||
model.save("ttt")
|
||||
|
||||
# model = word2vec.Word2Vec.load('ttt')
|
||||
# for sentence in sentences:
|
||||
# for word in sentence:
|
||||
# print(word)
|
||||
# if model[word].any():
|
||||
# print('get')
|
||||
print("done")
|
||||
|
||||
model = word2vec.Word2Vec.load('ttt')
|
||||
# data_preprocess('24444')
|
||||
print('finish')
|
||||
Loading…
Reference in New Issue