From 5701333e6b1b6a12b8a2f578bc1e7af7e36a8f62 Mon Sep 17 00:00:00 2001 From: FrankCM-ia <59652945+FrankCM-ia@users.noreply.github.com> Date: Mon, 26 Jul 2021 16:15:14 -0500 Subject: [PATCH 1/2] Eliminar dir _pycache_ --- .gitignore | 1 + .../__pycache__/AnalisisDatos.cpython-39.pyc | Bin 2251 -> 0 bytes .../__pycache__/LimpiezaDatos.cpython-39.pyc | Bin 2015 -> 0 bytes .../ReduccionComplejidad.cpython-39.pyc | Bin 1457 -> 0 bytes __pycache__/ExtraerTweets.cpython-39.pyc | Bin 1077 -> 0 bytes __pycache__/Ntweets.cpython-39.pyc | Bin 198 -> 0 bytes __pycache__/ProcesarTema.cpython-39.pyc | Bin 4786 -> 0 bytes __pycache__/Stanza.cpython-39.pyc | Bin 297 -> 0 bytes __pycache__/autenticate.cpython-39.pyc | Bin 578 -> 0 bytes 9 files changed, 1 insertion(+) create mode 100644 .gitignore delete mode 100644 Funciones/__pycache__/AnalisisDatos.cpython-39.pyc delete mode 100644 Funciones/__pycache__/LimpiezaDatos.cpython-39.pyc delete mode 100644 Funciones/__pycache__/ReduccionComplejidad.cpython-39.pyc delete mode 100644 __pycache__/ExtraerTweets.cpython-39.pyc delete mode 100644 __pycache__/Ntweets.cpython-39.pyc delete mode 100644 __pycache__/ProcesarTema.cpython-39.pyc delete mode 100644 __pycache__/Stanza.cpython-39.pyc delete mode 100644 __pycache__/autenticate.cpython-39.pyc diff --git a/.gitignore b/.gitignore new file mode 100644 index 00000000..5bc2625a --- /dev/null +++ b/.gitignore @@ -0,0 +1 @@ +_pycache_ \ No newline at end of file diff --git a/Funciones/__pycache__/AnalisisDatos.cpython-39.pyc b/Funciones/__pycache__/AnalisisDatos.cpython-39.pyc deleted file mode 100644 index 9cc2c318918260392c60c369cb8046cb005c6e96..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 2251 zcmZ9NTW{P%6vt;gw%5B!nxuqYC@KK~?F!m*Q@ON4MTtO2XoW~AXcfA+9&hTc*WS#` zHl><wJGfqMiuXgU^IrBUJbDS#Z!H zse^~?i}7E>U%!RIvUT>$xmjSD4bHPYX89hs!887hbG98?hIjK`H*j$}ndFspQC3eY z(-m&HsIo%lmC=KH_0B*RS%2Qt5AsS6%ub#g7w*82=S(FyWxtePZ2nm3G|evQ)%Db* zn;ZE!*GA>5-~PP$y{fZZr{e;zt5;>d`dwaaP18}n+Dpw;rRC;ErizK)yiIS`gI>O) zH@}|3fV#@{=2ul(7P`=QZ>_IReshc7$ib0?bYo2g;|V{*75UJ0QkC5=9@c9ccQV3q z_=IRg_TCP_+T-wQ zrXTL}8;_%@8D9Ae7=_L!F3GE`mPNG%55=Y)U{N=8Z3!$xm5y^4nR-%WF07~%rcx?T zu_+)s=AzqV_Z*s<>}7m~t5;EHoO%b(v75Y)LCI&_JN)9I!{PiT+=k;WxZ9^}#sy`IwEyKfummRG8i)mRl&W6wPN5%yO1mf46w0CY)7uCV`MOdPJ81WGP znCDAeUBK8e+sRu;Ug&$EUt7Sh0FfJmoYbUb0bn(NqoaWjEa(EI$yAdWH!-#sVvn!_=zf@mCbGcefkh^k zp^bI}4>h)8)UvUJg%bZnEwE_AM&_GU~k`{Xy1Y`l!(TWG$^s zAycXRaH8JCtfI7|E>f|I!i8ClkQHTKxk#neR<7Q_aJSXa4R&Nd0k45~NF;{?GgR;R zeU++r94}5X4$P3|^$H4tn?PSH0o8=}I#(ZI{1{-0D1z)G{tq-rR{#qgdCdjT93mcg z?m}Gx2{I%t5q>~!)aOxQPb7m`Y$L2)nYC<71`mW<2Tp<5inrIHJJKdMZ%2KbR=q8F{+v~K!O8meQiCQYZ z>=-^IC+0 z8y$Ym1cUb8-JZ7rK!v^k;i!*sPB&EVp=#h`s3!`=l6FI@~56Wei@jYf~h(NPB;yT+xFC@`P^`g+?%cmZ-iE@;1+0mPth%M zlUrAaTjB*?gtvK#+wdK{gLlGr@-E&D-^I(k2fmwcBSy(eQ|D6bZXD2e;I=szQ($%Z#2`YT@2kY32feHqjhdw*7lL{%$dRl)Myr9&%-tlOvyxFlcw|I>NJ&kqtaCUFRN}|A`pAoz{{tg)eT^uXZiu1bF!sGr zjUd?Eq>of0fwPssp2!#N)3)Kk_L0T*)A`Hl+7nu&QXjr0C(Q_*Q_hNE1e`Vjf@Vf) z>>!BErkQAF7~D$9irKVM6P)R77E){3NDI8MVl=Jk11Yj~ij~HzI3wME|F_os-PZi# z-Xq7358Yoox%>Xgz|m8!`5zBDgL~%QI>&aiJ!~)A$M&-WY=9kP4m-pSvmWaUZ=0&iY8{MA8eRg{ z?Zh3`i-8zc&Mb@e7&h&pQYX9xQq3(5cn;}71PPe}voIRm_{z|9SvI)|FadL7rN#mU z+-Nt)sx$_$|J?f0o%zA_rEhMmFJ0Hc?X~r#Z*}k;f~(62uC7fSM)3<5LT^&Z9eBto zNd1mkNs}lDMOHdnIX(LB*xRyCFPC*bq83jD=%!J?Rzxob(WG0>7prwINra3tQwsGY z79EeCJ%FP&x~rFVqraqHc`ExygxZC>nozmS^dd{$AV2|`Yl$Iwk2N$5 z_V@Lrdr14(1A$JGT(shJ6qFxeWuAFXK4J zl>6Y^o=pn0Q-4;e@-U<_VeL+$jm$!blhILeGzWmXg>m$EKB z4*9XHHw>0UH0R|io~u{hZ#nJq!sSThXFi7De{F0J*J9oX#fVNS7keA!#m@wL NExTkF?J`DO{{jjH=+FQF diff --git a/Funciones/__pycache__/ReduccionComplejidad.cpython-39.pyc b/Funciones/__pycache__/ReduccionComplejidad.cpython-39.pyc deleted file mode 100644 index 44d20282f1484a7452fcbc271a3ae3f5c234ec5b..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 1457 zcmZux&2Aev5ay7(tCcLtvhr6aMT!O}kPRFkf}Db&Xpy2PLp35J1q2uBMr4(>SGyaM zicw<)6i}}L`Uv(h$38>fg4djK=>zoC8A@>MAPbNj4tF`^*HRKj>rM;V^&TB$?o9h3}qx6Z^bB*v26ayMhz9n z7Wy{c@=+|?vV*>+?8;4CTe2s&aNUr7c@5XLYRKzL7KSLPrQu2t!B zqKqlEaY3FIQ`eXqHJ(=U2~dFbBXH$7p5ZWeak8vtOISb+M<8LF9q|RPxwpiSz2gH- zpC^F}jUDXmBS=+PRbVI=daN$=I1oZ)tj;j7kmbZd`i|)ye)`|z-<}-q?+s`?JUGd$ zRr>J9c`?b#LYc$oO3o)_KPzWdu8uO9%H8UB*Yvkq%(qEUBb*q-&<5Q{%QZ}u7MXd4 zO!cy9QED~5VDC!8gr1W236m}?JA;~?6ZJ4ixa+QOTJ!7zgM=rbJNT|oU^Q*w*2Q!- z!yGqpXNX4z`WDVtLXtng;X~d>!(yc^23^9E|G<8^z5vT>RwLU>^w#i{z3$d*$aIV^ zBaxxT`Y9)JRrqWXW#p@G9Auwj>Rk{uJ|KmdYk0|;uRioae~J4m_(=aWJiZrb5G4}M zQ3aAu0qYMnzr_|XWeZ`s6joedzq<$!aj=c3-++Eli?j9q9QFMAfemYcY7#YSCkY0c zusE*mN<@(+YV-z8U(mEo$5C2Us*tW@Dn0(GPh)_%c($j_>TKXP(uaHEVxghBYZjcQ4 zq>%R(oXT2rDx~DcLicMpqtc?s6M}mrLadtTgYP8qF?CH1md&-ASpdx$)$?zW0!hcg^|_qbXMx z?`M_DvqEWVIIlZ>gDldw7m+8)ht;Hi2QOV%VaMA`wEF_AdgbY?l=EDDOIaH#H+642 SM7qN^#6cXz?YI#m`Tqlsd_blE diff --git a/__pycache__/ExtraerTweets.cpython-39.pyc b/__pycache__/ExtraerTweets.cpython-39.pyc deleted file mode 100644 index ad2a0f84968a193b73b15a968497a4c2d64684aa..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 1077 zcmZWnOK%iM5bmDo*_j=$7r(HxA;6MLv{7WkHL?&9jv^#vA(<$owPsc-eI7zMk7RI0c^lf2 z^azh3mMzpv(1B$c%IF1wr9CpEj(8WklFIh49%gjm0(57T5*hOq=maNL7cxh%W#!YP}74l8FAt3KeD zU|p_k5x6w|Laz467LjXM|1w;G%W{3@{r)At3Uu6eakwfscfjE6tqgCMW^j$S+&yfCH5p!|`4TP!WB8H*<@&J9t%&>I-MeFbQD?lf zKs~8*kpd4F-VMtnqNM!hhMUX!8jKAb%-ROE%1UFyaxsqepTaRF7{$X>s2Ysm(Wn`) z(w5c2Jh4HhWu|OUJawvdd#ZGWT7yujiyA1fZd1<&iR+)Y-DAmNmLmvN4W(rZTTYAm z-`y^(&@~$EB4X1VZn{56e}f{TjyiseuJ|#H{K((*J2dhba)&yKY3yI|HqrYBIYcR! diff --git a/__pycache__/Ntweets.cpython-39.pyc b/__pycache__/Ntweets.cpython-39.pyc deleted file mode 100644 index 484c2ea08c7b760324d1c5fccb15df0ab34e0bcd..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 198 zcmYe~<>g`k0;j#N6L^92V-N=!FakLaKwQiLBvKfn7*ZI688n$*rU1EqnvA#D^GeE7 zQ%j0hG88cbg}}tGm=vp+(7fWr#ANki1((E<#F*gJ+|=TdqErRXh?sz){N&W)#N5o( zypnu{lvD+u)V%c4#H>^Wzr>Q#qQsn-;N+srg5sDEu*va$spZ8neh_Q)3My}L*yQG? Ol;)(`f!y{Ph#3HKFg9BN diff --git a/__pycache__/ProcesarTema.cpython-39.pyc b/__pycache__/ProcesarTema.cpython-39.pyc deleted file mode 100644 index d4a3ea3e55d305af395a4d5e064bc2ecb0881a84..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 4786 zcmc&&O>7)V74GW4>G}0foH)*}Hz3GtSSMKlD;5)WH(4W4Vj^rdqDeH{>FROX?&%&^ z)!^9b5eMf;wBo{n(`0Yqz!7Qh9QMqy#SJdGA#uxz?^Vw@wv!x?keE@wSHC^4-uvE9 zP0(!C41B)&+rK{lYThvZL7l@N3!RVf$bW)xgR{i&OU`_z`=)PdZuvI2nK(VycNwj* zl1k6>J>6zWwO8}&x@{+oUej;twv)8hX|0n?jrp|ZZZgxG^=HR>PxvRI6Qem^dB!l} z@#+)9Kgnyn4t|O^coTe{xA+wJX+F(oz!&%|KLLJ*&+(JsXZa~U4}OlH<_qBG`5Ar| z`~tto&+`jUn17M8HKV=w0w*)3#o41SyzZ+W0 z%&kpppM8-IZDo&~+$=w5mR8 zV1(r?ca{y+B)(wmkNfwadmG<UJ?aHZ{oSG#CQ*>;m(P$v&w&`sW)0@ysezW?7I=%TGO>vH(4T#Foxy<9=;yJ-a|XLfQdiD>UPeyOmP`j`5Nn7V8j_?q@pnb-@qzIW5WYBJYb`m zS4S0~psFU=r~(^RRm-b1t4AFmrW~a;?xY9&)G-c!dkq7BI)VWc>wqt21d%x+vmk{V^Z`d+n0+Ad7ErPx zf?m|Fi_U(Llqak0IysObul8dyq!n+zEm(Y># zf|LNUH8gAumI+Wwn6%hsK<6@>X5u?oeI$GkbSNC_@Ifz|KxN1{S-snS6_RTjW?{}(ZJm>a^t}qMWdYUBd%}R0(kqqxoxi*&)Ez5X=vO^qV&00m`m5- zLKGX1pSQN0#8QP>uYco**z#{Y#@?m%aTx8~j(YH&cElukY*3}x9uGG zf6i9hrr)IOv4*rkiCy~B2cYU`ab(*pW)z`Vb`H`=D9+>8j?3Tb>og%dVJ7+mxq>C* z19))iStf2^4qb3Ze$W6NM{1f>VFsAC{C&S1kJCS<>qvyq9+@~ z?(DOD)Id0$tDH?2JCKi%BPw}CR~knYM?vWzh#ig-IRecLPd6X(y*jv%AEF2M} z8WAe0 zCFUh514gvSh&2nq!*iN~*3U2=QjnT_NPh>$ypIltndVW(&%;;1EdUNlf!o~S?o*RW z)lGBL;uY>aWo1!1LZYDbubh#~b-|!2yf)d*>+>)tmGLlOV|B0k6A^UdFpyVYva5Uc zQa{}&oSjFpie8Xy;AqnigK%p@K$#p)lw>F?d%@!lX+&yULi2X;IF^N#b-R1c>fm9t zxK~-cDWZS}&@7#)SlEctNCYbCbVb$!iQpG7O(YApPUU2;sHm*djT01?{X~how9Dy> zI8|-Cs3LKzE;Qanxrv&9@*QeQ-_ce78w@~!L(gsCJ(B7cgUzvX@F4imuhBaKT@sPZ zpnZs}Iq`XT3Gy#|oP5QEr=aSe__$S=s!Mq2B57&aw4f+kx>UCKHOPz*+=t+PebUBi zzcuNUyN*kL3f1i^4efSto;`$Bv;3VeK;-k2r*fqP#5E zXAVWIGSX^C@uzWwBW`m15C?Hr+v)o5Yr>OzuY{!d$%hI=>T4tX5eXON*(Z5tLtb z>FGDSvFa#9FWJpR&!5@ARjC^VY9OKxxrFrLK!|iQq8dv7!mIc;B?9`j@lA}Di_0`% zQ;i_VNBqY6O6RAyRywyoy>sVdzdr7*t$%vQ*WRbEYrb3NQQs}+3;Q=wC?VTnK?yA%CRaBhu&T{g+Kc5KLth{WIE4#q z@IaO^U6+-VkNi4aPM1odOO4NhRikp4{2k;|W7!QB`4 z-bgOpj(hz$+6$14GU;D{4v53CmOvH_FX!m1;W#C?QVZAPUEnh*^hkm~CI?oZ diff --git a/__pycache__/autenticate.cpython-39.pyc b/__pycache__/autenticate.cpython-39.pyc deleted file mode 100644 index 542ee2b111dc8b97771e149cd6d54501e3957362..0000000000000000000000000000000000000000 GIT binary patch literal 0 HcmV?d00001 literal 578 zcmYjPzi-n(6uz?^6oCs!urM-4A}Fz4*G33IW59tlDa1)*t5R@%M`QU{b9bqm$lmxv z+L48o|AfIS6B{!V=c=H*)A#9p_kP@y-uXK_w*j|rKYoAu4gvUCi)|CQc*#K@azX&X z7(W~!NXH%eFQXr`eWirUo_UESN&jrRVbn~4ypGR}p{X93 zot9YLVmZcV>q;KITCs)grzpmxx=%5i`$2#yowF>$X{%AmL6*{0g2_C>>+0TrMR7nd ztK{wU>cO^}-=s)Bf8r}3_458a{`~Cta4@XVcJh!hOpbXRrz{M7hM)48ny2*?yhf}y sa+ec;(#?^f42jHdS^}xpeeGX&-l97^P&J Date: Mon, 26 Jul 2021 16:45:54 -0500 Subject: [PATCH 2/2] Eliminar dir Exp --- Exp/Exp.ipynb | 231 -------------------------------------------------- Exp/Exp.py | 79 ----------------- 2 files changed, 310 deletions(-) delete mode 100644 Exp/Exp.ipynb delete mode 100644 Exp/Exp.py diff --git a/Exp/Exp.ipynb b/Exp/Exp.ipynb deleted file mode 100644 index ae0120bf..00000000 --- a/Exp/Exp.ipynb +++ /dev/null @@ -1,231 +0,0 @@ -{ - "metadata": { - "language_info": { - "codemirror_mode": { - "name": "ipython", - "version": 3 - }, - "file_extension": ".py", - "mimetype": "text/x-python", - "name": "python", - "nbconvert_exporter": "python", - "pygments_lexer": "ipython3", - "version": "3.9.5" - }, - "orig_nbformat": 4, - "kernelspec": { - "name": "python3", - "display_name": "Python 3.9.5 64-bit" - }, - "interpreter": { - "hash": "4f1212124a69d0e192309562875881c403391e14c64cb7bbd7e67cd7a8d12ddb" - } - }, - "nbformat": 4, - "nbformat_minor": 2, - "cells": [ - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "# import tweepy\n", - "# from tweepy import api\n", - "# from tweepy import streaming\n", - "# from tweepy import auth\n", - "# from tweepy.streaming import Stream\n", - "# from autenticate import get_auth\n", - "\n", - "# class TweetsListener(tweepy.StreamListener):\n", - "# def on_connect(self):\n", - "# print(\"Estoy conectado\")\n", - "\n", - "# def on_status(self, status):\n", - "# print(status.text)\n", - "\n", - "# def on_error(self, status_code):\n", - "# print(\"Error\", status_code)\n", - "\n", - "# auth = get_auth()\n", - "# api = tweepy.API(auth, wait_on_rate_limit_notify=True , wait_on_rate_limit=True)\n", - "\n", - "# stream = TweetsListener()\n", - "# streamingApi = tweepy.Stream(auth=api.auth, listener=stream)\n", - "\n", - "# streamingApi.filter(\n", - "# track=[\"pedro castillo\", \"keiko\"]\n", - "# )\n", - "# import nltk\n", - "# nltk.download('stopwords')\n", - "\n", - "#print(\"frecuente\\xa0hsy\")" - ] - }, - { - "cell_type": "code", - "execution_count": 152, - "metadata": {}, - "outputs": [ - { - "output_type": "execute_result", - "data": { - "text/plain": [ - " going to today i am it is rain\n", - "1 1 1 1 0 0 1 1 1\n", - "2 1 0 1 1 1 0 0 0\n", - "3 1 1 0 1 1 0 0 0" - ], - "text/html": "
\n\n\n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n
goingtotodayiamitisrain
111100111
210111000
311011000
\n
" - }, - "metadata": {}, - "execution_count": 152 - } - ], - "source": [ - "# creacion del primer data frame\n", - "dic1 = {'going': [1, 1, 1],\n", - " 'to': [1, 0, 1],\n", - " 'today': [1, 1, 0],\n", - " 'i': [0, 1, 1],\n", - " 'am': [0, 1, 1],\n", - " 'it': [1, 0, 0],\n", - " 'is': [1, 0, 0],\n", - " 'rain': [1, 0, 0]}\n", - "docs = [1,2,3]\n", - "df1 = pd.DataFrame(dic1, index=docs)\n", - "df1" - ] - }, - { - "cell_type": "code", - "execution_count": 183, - "metadata": {}, - "outputs": [], - "source": [ - "import pandas as pd\n" - ] - }, - { - "cell_type": "code", - "execution_count": 184, - "metadata": {}, - "outputs": [ - { - "output_type": "stream", - "name": "stdout", - "text": [ - " count\ngoing 3\nto 2\ntoday 2\ni 2\nam 2\nit 1\nis 1\nrain 1\n" - ] - } - ], - "source": [ - "#print(df2.shape)\n", - "df3= count_col(df1)\n", - "print(df3)" - ] - }, - { - "cell_type": "code", - "execution_count": 163, - "metadata": {}, - "outputs": [], - "source": [ - "def WordsXDocuments(dataframe):\n", - " WD = pd.DataFrame() \n", - " col_list= list(dataframe)\n", - " dataframe[col_list].sum(axis=1)\n", - " WD['Cantidad'] = dataframe[col_list].sum(axis=1)\n", - " print(WD)\n", - "\n", - " df_tf = dataframe.copy()\n", - " for col in df_tf:\n", - " nrow = len(df_tf[col])\n", - " for i in range(0,nrow):\n", - " print(WD.loc[i+1,'Cantidad'])\n", - " df_tf.loc[i+1,col] = df_tf.loc[i+1,col] / WD.loc[i+1,'Cantidad']\n", - " return df_tf\n", - "\n" - ] - }, - { - "cell_type": "code", - "execution_count": 164, - "metadata": {}, - "outputs": [ - { - "output_type": "stream", - "name": "stdout", - "text": [ - " going to today i am it is rain\n1 0.166667 0.166667 0.166667 0.00 0.00 0.166667 0.166667 0.166667\n2 0.250000 0.000000 0.250000 0.25 0.25 0.000000 0.000000 0.000000\n3 0.250000 0.250000 0.000000 0.25 0.25 0.000000 0.000000 0.000000\n" - ] - } - ], - "source": [ - "WD = WordsXDocuments(df1)\n", - "print(WD)" - ] - }, - { - "cell_type": "code", - "execution_count": 173, - "metadata": {}, - "outputs": [ - { - "output_type": "execute_result", - "data": { - "text/plain": [ - " idf_values\n", - "going 0.00\n", - "to 0.41\n", - "today 0.41\n", - "i 0.41\n", - "am 0.41\n", - "it 1.09\n", - "is 1.09\n", - "rain 1.09" - ], - "text/html": "
\n\n\n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n \n
idf_values
going0.00
to0.41
today0.41
i0.41
am0.41
it1.09
is1.09
rain1.09
\n
" - }, - "metadata": {}, - "execution_count": 173 - } - ], - "source": [ - "dic2 = {'idf_values': [0, 0.41, 0.41, 0.41, 0.41, 1.09, 1.09, 1.09]}\n", - "words = ['going', 'to', 'today', 'i', 'am', 'it', 'is','rain']\n", - "df2 = pd.DataFrame(dic2, index=words)\n", - "df2 " - ] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [ - "def tf_idf(df_tf, df_idf):\n", - " df_tf_idf = df_tf.copy()\n", - " for col in df_tf_idf:\n", - " nrow = len(df_tf_idf[col])\n", - " for i in range(0,nrow):\n", - " df_tf_idf.loc[i+1,col] *= df_idf.loc[col,'idf_values']\n", - " return df_tf_idf\n" - ] - }, - { - "cell_type": "code", - "execution_count": 167, - "metadata": {}, - "outputs": [], - "source": [] - }, - { - "cell_type": "code", - "execution_count": null, - "metadata": {}, - "outputs": [], - "source": [] - } - ] -} \ No newline at end of file diff --git a/Exp/Exp.py b/Exp/Exp.py deleted file mode 100644 index 0e30065c..00000000 --- a/Exp/Exp.py +++ /dev/null @@ -1,79 +0,0 @@ -import numpy as np -import pandas as pd -import re - -#Visualización -import matplotlib.pyplot as plt -import matplotlib -from wordcloud import WordCloud, STOPWORDS - -#nltk librería de análisis de lenguaje -import nltk -#Este proceso puede hacerse antes de forma manual, descargar las stopwords de la librería nltk -nltk.download('stopwords') -from nltk.corpus import stopwords -stop_words_sp = set(stopwords.words('spanish')) -stop_words_en = set(stopwords.words('english')) -#Concatenar las stopwords aplicándose a una cuenta que genera contenido en inglés y español -stop_words = stop_words_sp | stop_words_en -from nltk import tokenize - -matplotlib.style.use('ggplot') -pd.options.mode.chained_assignment = None - -#Últimos 400 tweets previamente descargados -tweets = pd.read_csv('sample_tweets-400.csv') -#Últimos 3240 tweets previamente descargados -tweets2 = pd.read_csv('sample_tweets.csv') - -def wordcloud(tweets,col,idgraf): - #Crear la imagen con las palabras más frecuentes - wordcloud = WordCloud(background_color="white",stopwords=stop_words,random_state = 2016).generate(" ".join([i for i in tweets[col]])) - #Preparar la figura - plt.figure(num=idgraf, figsize=(20,10), facecolor='k') - plt.imshow(wordcloud) - plt.axis("off") - plt.title("Good Morning Datascience+") - - -def tweetprocess(tweets,idgraf): - #Monitorear que ha ingresado a procesar el gráfico - print(idgraf) - #Imprimir un tweet que sepamos contenga RT @, handles y puntuación para ver su eliminación - print(tweets['text'][3]) - tweets['tweetos'] = '' - - #add tweetos first part - for i in range(len(tweets['text'])): - try: - tweets['tweetos'][i] = tweets['text'].str.split(' ')[i][0] - except AttributeError: - tweets['tweetos'][i] = 'other' - - #Prepocesar tweets con 'RT @' - for i in range(len(tweets['text'])): - if tweets['tweetos'].str.contains('@')[i] == False: - tweets['tweetos'][i] = 'other' - - # Remover URLs, RTs, y twitter handles - for i in range(len(tweets['text'])): - tweets['text'][i] = " ".join([word for word in tweets['text'][i].split() - if 'http' not in word and '@' not in word and '<' not in word and 'RT' not in word]) - #Monitorear que se removieron las menciones y URLs - print("------Después de remover menciones y URLs --------") - print(tweets['text'][3]) - - #Remover puntuación, se agregan símbolos del español - tweets['text'] = tweets['text'].apply(lambda x: re.sub('[¡!@#$:).;,¿?&]', '', x.lower())) - tweets['text'] = tweets['text'].apply(lambda x: re.sub(' ', ' ', x)) - #Monitorear que se removió la puntuación y queda en minúsculas - print("------Después de remover signos de puntuación y pasar a minúsculas--------") - print(tweets['text'][3]) - #hacer el análisis de WordCloud - wordcloud(tweets,'text',idgraf) - -#Graficar tendencia 400 tweets -tweetprocess(tweets,100) -#Graficar tendencia 3240 tweets -tweetprocess(tweets2,200) -plt.show() \ No newline at end of file