From daa568709b16ce3798b8f186f4604e20300b376e Mon Sep 17 00:00:00 2001 From: Jared Bell Date: Mon, 14 Sep 2026 19:00:07 -0400 Subject: [PATCH] first commit --- .idea/.gitignore | 10 + .idea/anya.iml | 15 + .../inspectionProfiles/profiles_settings.xml | 6 + .idea/misc.xml | 4 + .idea/modules.xml | 8 + .idea/vcs.xml | 6 + __pycache__/main.cpython-314.pyc | Bin 0 -> 4673 bytes main.py | 72 ++ resources/sources.txt | 20 + resources/stopwords.txt | 639 ++++++++++++++++++ .../__pycache__/headlines.cpython-314.pyc | Bin 0 -> 6037 bytes .../__pycache__/normalization.cpython-314.pyc | Bin 0 -> 3615 bytes .../__pycache__/similarity.cpython-314.pyc | Bin 0 -> 5047 bytes services/__pycache__/sources.cpython-314.pyc | Bin 0 -> 2162 bytes services/headlines.py | 94 +++ services/normalization.py | 56 ++ services/similarity.py | 57 ++ services/sources.py | 31 + structs/__pycache__/headline.cpython-314.pyc | Bin 0 -> 2815 bytes structs/headline.py | 32 + tests/__init__.py | 0 tests/__pycache__/__init__.cpython-314.pyc | Bin 0 -> 149 bytes ...est_headlines.cpython-314-pytest-9.0.3.pyc | Bin 0 -> 9036 bytes .../test_headlines.cpython-314.pyc | Bin 0 -> 5146 bytes tests/test_headlines.py | 154 +++++ 25 files changed, 1204 insertions(+) create mode 100644 .idea/.gitignore create mode 100644 .idea/anya.iml create mode 100644 .idea/inspectionProfiles/profiles_settings.xml create mode 100644 .idea/misc.xml create mode 100644 .idea/modules.xml create mode 100644 .idea/vcs.xml create mode 100644 __pycache__/main.cpython-314.pyc create mode 100644 main.py create mode 100644 resources/sources.txt create mode 100644 resources/stopwords.txt create mode 100644 services/__pycache__/headlines.cpython-314.pyc create mode 100644 services/__pycache__/normalization.cpython-314.pyc create mode 100644 services/__pycache__/similarity.cpython-314.pyc create mode 100644 services/__pycache__/sources.cpython-314.pyc create mode 100644 services/headlines.py create mode 100644 services/normalization.py create mode 100644 services/similarity.py create mode 100644 services/sources.py create mode 100644 structs/__pycache__/headline.cpython-314.pyc create mode 100644 structs/headline.py create mode 100644 tests/__init__.py create mode 100644 tests/__pycache__/__init__.cpython-314.pyc create mode 100644 tests/__pycache__/test_headlines.cpython-314-pytest-9.0.3.pyc create mode 100644 tests/__pycache__/test_headlines.cpython-314.pyc create mode 100644 tests/test_headlines.py diff --git a/.idea/.gitignore b/.idea/.gitignore new file mode 100644 index 0000000..30cf57e --- /dev/null +++ b/.idea/.gitignore @@ -0,0 +1,10 @@ +# Default ignored files +/shelf/ +/workspace.xml +# Editor-based HTTP Client requests +/httpRequests/ +# Ignored default folder with query files +/queries/ +# Datasource local storage ignored files +/dataSources/ +/dataSources.local.xml diff --git a/.idea/anya.iml b/.idea/anya.iml new file mode 100644 index 0000000..5328eca --- /dev/null +++ b/.idea/anya.iml @@ -0,0 +1,15 @@ + + + + + + + + + + + + + + + \ No newline at end of file diff --git a/.idea/inspectionProfiles/profiles_settings.xml b/.idea/inspectionProfiles/profiles_settings.xml new file mode 100644 index 0000000..105ce2d --- /dev/null +++ b/.idea/inspectionProfiles/profiles_settings.xml @@ -0,0 +1,6 @@ + + + + \ No newline at end of file diff --git a/.idea/misc.xml b/.idea/misc.xml new file mode 100644 index 0000000..e91b2f2 --- /dev/null +++ b/.idea/misc.xml @@ -0,0 +1,4 @@ + + + + \ No newline at end of file diff --git a/.idea/modules.xml b/.idea/modules.xml new file mode 100644 index 0000000..0ffd615 --- /dev/null +++ b/.idea/modules.xml @@ -0,0 +1,8 @@ + + + + + + + + \ No newline at end of file diff --git a/.idea/vcs.xml b/.idea/vcs.xml new file mode 100644 index 0000000..94a25f7 --- /dev/null +++ b/.idea/vcs.xml @@ -0,0 +1,6 @@ + + + + + + \ No newline at end of file diff --git a/__pycache__/main.cpython-314.pyc b/__pycache__/main.cpython-314.pyc new file mode 100644 index 0000000000000000000000000000000000000000..346100e34dca481d827325c5bc909d59110d307c GIT binary patch literal 4673 zcmbtYT~Hg>6~3$0FA_o$0vl}5Vwo5@ST+F%?7CoVASkGSB{{Affx2iHv9XZk-4#E2 z=#c45il6LE(urr{2R}43?zE3|hKFRvP2DjQXuHI+QMY*N8MhC8BQ^b*`la`-v_IIS zB<;><@40*azH{!mchBlodu0WHS2cI%ojW!Fo}!6-8Op@7dyrTF;~<0u0FOpp02|ct zIuD}p06wVa^@9f9AkFmw{)PySXJr&(NtfiKb&{UM5cy%`MY3$B{|MfQ}53oM@ax4*2{&|8bV)N+yq&M?XKu z=j(PbWMGJuCW60F0 z5f!yeBNZsEJ7?MH7ikdXh=zSehb8{xorkETNXeOtXD|uBT z%b$#7crsNMg&0&xat4M;PQXONP6;M^8n$cH&bJ$iNSIUiB@Zd z(@yg}VP+aez%u+i!6h;*O-#`IEKR48_L!6*iC{RQ+21hOdyhU;Y(IRtaBqOD|M9`%AJfK)oEWkBE%BP8Z zpr2)7s>?Z5Q=#CN7Y7B~(#-m#7+!oR<$YAii^dEhQ*d8VexiSdOy%H)ld2r83@6zt zg#oRrJCaP%EYJCUPSKdkB$G5N>QPCH28y1@B}Hp^=*6f$jIoWkx*Y) znzN{b74a0E7A^gA2|A0)LNq7fioo5TV$l)JqLWQDuw*((i_Q{QwdLnT(=5rN@)2zm z!(~(Ce4MA}c+rrBbq5C;Wz1J0n#$W)#Awl?>@m?ICx;WAQh~+QbZ}yo5+!_`gXZii zH_lLVBF2c=JEEh!=CF?9(gJg+Pa>2{G)mfmQn+BDm*Td!FKcgNC|g8AI0tKrs_ z2d<-wCY5nNxs+HwvXm0Sk%IZ+R)topZK)e+xto6Kf8cIgv?yAy|K9g0z8v_}E<`UE z%x~}18d$?u-dMFkYn$%o+sOy+QyR^owXksE^6GfO9NVq{0W65E1Xd+8gO_hr-;93T z_>1;C^FQxi?tS>`cOTYu-)(!ezfbr9Dd=mI;A6sgrl8M0sol3QQmm<67~ZUOeb|0u z{)6sEm4_CwO=sU1?RTu2l*9yYE?7E(p=F#j1jN{IS{oe9HjY&$k^fNuIsu7eZlSd;*Ck z{=4YirWFc_g*9SFfx7|vhIvv_GF5D7ywmiFZG$+qPMlh9Tfsg{ZFFB+@4f_!LZF!~ z95@^P2{7js?3r|bzJ((L|K=!I(7&?wy52klK5y)G4ebZNZtJxK?cn|^=jww!`Y#Sy zhidRIjx`K9@KuKa(yRNeK`XxMZ3r6iHKPI2Yj$h!48B&kKiG+{bs7-eV;y#xep_*7 zxYG2yN&}=tb3BfAQ9SN*h$eL6pv~{%Xm*ByuVBBjp-Trxt-_H1FOc3nrK1Hk^Kkdd zs^uf1bShaVNRCOIms$wX(AR(N;v1p`{<{z?opBHa`uiXrXVG^siyBh43f<7={uy`% z?|}8fn`k~WouP6mx|=-#r|8?0i^JO%j$zmzf&DQ!@h8yt7_>YArVUX2C8#b|IIkbO zcC2VU`*#D5>9)+E!nt9o|I$)_v+IH7@PhG){m6o~=xklE|IOHrwQZ#^ptmjN3wYfh R^yc>lV7ck*+2hd1e*lK+4^sdD literal 0 HcmV?d00001 diff --git a/main.py b/main.py new file mode 100644 index 0000000..d788e5a --- /dev/null +++ b/main.py @@ -0,0 +1,72 @@ +import logging +from services.headlines import prepare_headlines +from services.normalization import get_stopwords, normalize_headline +from services.sources import get_sources + +logging.basicConfig( + level=logging.INFO, + format='%(asctime)s [%(levelname)s] [%(name)s]: %(message)s' +) +logger = logging.getLogger(__name__) + +SOURCE_FILE = './resources/sources.txt' +STOPWORDS_FILE = './resources/stopwords.txt' +SIMILARITY_THRESHOLD = 0.75 + + +def main(): + logger.info("Starting duplicate headline detection application") + logger.debug("Source file configured at: '%s'", SOURCE_FILE) + logger.debug("Stopwords file configured at: '%s'", STOPWORDS_FILE) + + try: + logger.info("Loading news sources from '%s'", SOURCE_FILE) + sources = get_sources(SOURCE_FILE) + logger.info("Loaded %d sources successfully", len(sources) if sources else 0) + except Exception as e: + logger.critical("Failed to load sources from '%s': %s", SOURCE_FILE, e, exc_info=True) + return + + try: + logger.info("Loading stopwords from '%s'", STOPWORDS_FILE) + stopwords = get_stopwords(STOPWORDS_FILE) + logger.info("Loaded %d stopwords successfully", len(stopwords) if stopwords else 0) + except Exception as e: + logger.critical("Failed to load stopwords from '%s': %s", STOPWORDS_FILE, e, exc_info=True) + return + + try: + logger.info("Fetching and preparing headlines from %d sources", len(sources)) + headlines = prepare_headlines(sources, stopwords) + logger.info("Total prepared headlines available for comparison: %d", len(headlines)) + except Exception as e: + logger.critical("Failed during headline preparation: %s", e, exc_info=True) + return + + total_comparisons = (len(headlines) * (len(headlines) - 1)) // 2 if len(headlines) > 1 else 0 + logger.info("Beginning pairwise headline comparisons (total comparisons to execute: %d)", total_comparisons) + + duplicate_count = 0 + comparison_idx = 0 + for i in range(len(headlines)): + for j in range(i + 1, len(headlines)): + comparison_idx += 1 + logger.debug("Comparison [%d/%d]: Headline %d vs Headline %d", comparison_idx, total_comparisons, i, j) + try: + similarity_score = headlines[i].compare_headlines(headlines[j]) + logger.debug("Similarity score between [%d] and [%d] is %.4f (threshold: %f)", i, j, similarity_score, SIMILARITY_THRESHOLD) + if similarity_score >= SIMILARITY_THRESHOLD: + duplicate_count += 1 + logger.warning("Duplicate/similar headline match found (score: %.4f < %f): '%s' vs '%s'", + similarity_score, SIMILARITY_THRESHOLD, headlines[i].display_text, headlines[j].display_text) + print(f"Duplicate headlines found: {headlines[i].display_text}") + except Exception as e: + logger.error("Error during comparison between headline %d (%r) and headline %d (%r): %s", + i, headlines[i].display_text, j, headlines[j].display_text, e, exc_info=True) + + logger.info("Headline comparison completed. Evaluated %d pairs and found %d duplicate alerts.", + comparison_idx, duplicate_count) + + +if __name__ == '__main__': + main() diff --git a/resources/sources.txt b/resources/sources.txt new file mode 100644 index 0000000..0c0197b --- /dev/null +++ b/resources/sources.txt @@ -0,0 +1,20 @@ +https://www.cnn.com/us +https://www.cnn.com/politics +https://www.foxnews.com/us +https://www.foxnews.com/politics +https://www.nbcnews.com/us-news +https://www.nbcnews.com/politics +https://www.reuters.com/world/us/ +https://apnews.com/us-news +https://apnews.com/politics +https://www.npr.org/sections/national/ +https://www.npr.org/sections/politics/ +https://www.cbsnews.com/us/ +https://www.cbsnews.com/politics/ +https://www.forbes.com/business/ +https://www.ms.now/ +https://www.nytimes.com/section/us +https://www.nytimes.com/section/politics +https://www.usnews.com/news +https://time.com/ +https://www.axios.com/politics-policy \ No newline at end of file diff --git a/resources/stopwords.txt b/resources/stopwords.txt new file mode 100644 index 0000000..4eb985d --- /dev/null +++ b/resources/stopwords.txt @@ -0,0 +1,639 @@ +a +a's +able +about +above +abroad +according +accordingly +across +actually +adj +after +afterwards +again +against +ago +ahead +ain't +all +allow +allows +almost +alone +along +alongside +already +also +although +always +am +amid +amidst +among +amongst +an +and +another +any +anybody +anyhow +anyone +anything +anyway +anyways +anywhere +apart +appear +appreciate +appropriate +are +aren't +around +as +aside +ask +asking +associated +at +available +away +awfully +back +backward +backwards +be +became +because +become +becomes +becoming +been +before +beforehand +begin +behind +being +believe +below +beside +besides +best +better +between +beyond +both +brief +but +by +c'mon +c's +came +can +can't +cannot +cant +caption +cause +causes +certain +certainly +changes +clearly +co +co. +com +come +comes +concerning +consequently +consider +considering +contain +containing +contains +corresponding +could +couldn't +course +currently +dare +daren't +definitely +described +despite +did +didn't +different +directly +do +does +doesn't +doing +don't +done +down +downwards +during +each +edu +eg +eight +eighty +either +else +elsewhere +end +ending +enough +entirely +especially +et +etc +even +ever +evermore +every +everybody +everyone +everything +everywhere +ex +exactly +example +except +fairly +far +farther +few +fewer +fifth +first +five +followed +following +follows +for +forever +former +formerly +forth +forward +found +four +from +further +furthermore +get +gets +getting +given +gives +go +goes +going +gone +got +gotten +greetings +had +hadn't +half +happens +hardly +has +hasn't +have +haven't +having +he +he'd +he'll +he's +hello +help +hence +her +here +here's +hereafter +hereby +herein +hereupon +hers +herself +hi +him +himself +his +hither +hopefully +how +how's +howbeit +however +hundred +i +i'd +i'll +i'm +i've +ie +if +ignored +immediate +in +inasmuch +inc +inc. +indeed +indicate +indicated +indicates +inner +inside +insofar +instead +into +inward +is +isn't +it +it'd +it'll +it's +its +itself +just +k +keep +keeps +kept +know +known +knows +last +lately +later +latter +latterly +least +less +lest +let +let's +like +liked +likely +likewise +little +look +looking +looks +low +lower +ltd +made +mainly +make +makes +many +may +maybe +mayn't +me +mean +meantime +meanwhile +merely +might +mightn't +mine +minus +miss +more +moreover +most +mostly +mr +mrs +much +must +mustn't +my +myself +name +namely +nd +near +nearly +necessary +need +needn't +needs +neither +never +neverf +neverless +nevertheless +new +next +nine +ninety +no +no-one +nobody +non +none +nonetheless +noone +nor +normally +not +nothing +notwithstanding +novel +now +nowhere +obviously +of +off +often +oh +ok +okay +old +on +once +one +one's +ones +only +onto +opposite +or +other +others +otherwise +ought +oughtn't +our +ours +ourselves +out +outside +over +overall +own +particular +particularly +past +per +perhaps +placed +please +plus +possible +presumably +probably +provided +provides +que +quite +qv +rather +rd +re +really +reasonably +recent +recently +regarding +regardless +regards +relatively +respectively +right +round +said +same +saw +say +saying +says +second +secondly +see +seeing +seem +seemed +seeming +seems +seen +self +selves +sensible +sent +serious +seriously +seven +several +shall +shan't +she +she'd +she'll +she's +should +shouldn't +since +six +so +some +somebody +someday +somehow +someone +something +sometime +sometimes +somewhat +somewhere +soon +sorry +specified +specify +specifying +still +sub +such +sup +sure +t's +take +taken +taking +tell +tends +th +than +thank +thanks +thanx +that +that'll +that's +that've +thats +the +their +theirs +them +themselves +then +thence +there +there'd +there'll +there're +there's +there've +thereafter +thereby +therefore +therein +theres +thereupon +these +they +they'd +they'll +they're +they've +thing +things +think +third +thirty +this +thorough +thoroughly +those +though +three +through +throughout +thru +thus +till +to +together +too +took +toward +towards +tried +tries +truly +try +trying +twice +two +un +under +underneath +undoing +unfortunately +unless +unlike +unlikely +until +unto +up +upon +upwards +use +used +useful +uses +using +usually +v +value +various +versus +very +via +viz +vs +want +wants +was +wasn't +way +we +we'd +we'll +we're +we've +welcome +well +went +were +weren't +what +what'll +what's +what've +whatever +when +when's +whence +whenever +where +where's +whereafter +whereas +whereby +wherein +whereupon +wherever +whether +which +whichever +while +whilst +whither +who +who'd +who'll +who's +whoever +whole +whom +whomever +whose +why +why's +will +willing +wish +with +within +without +won't +wonder +would +wouldn't +yes +yet +you +you'd +you'll +you're +you've +your +yours +yourself +yourselves +zero \ No newline at end of file diff --git a/services/__pycache__/headlines.cpython-314.pyc b/services/__pycache__/headlines.cpython-314.pyc new file mode 100644 index 0000000000000000000000000000000000000000..3696f1059ce95b2a90e8e08a0f87c9b11ce2f2cf GIT binary patch literal 6037 zcmb6dTWk~A^^V`K*d{M0B#=o$0uCX84KG0;z><_CPB4z~Zosjb*b~RW@eDI#NW5&D zwjWK}50-8v*#1D(A6DweR+TEGt}3M6uA*I7O^7Rp*=(U|RcY0aW<}H&Qq^-kGq&+& zS@uevxxV+j&$;KGGdt~N76jwQ8+WEIZbIlU_>VfY%H-))n9QLN@)7+=z;8`IaY`d- z8ss&tpoKMUzwVS?(C5~4{f1LU!PtPtkk_e_E4&-gn0YTD{}(JFi(n0D1)I<4)4NcD z5X#1^U4+k&o7;UvGjf~0z`(<9LNZRolTj*>fN6P>;ijoXJWYouX)2nCCuvwUol@oz zO@SfO!yK#g^nbv14h7J5Wi#qRs2&9sVG!&D^LTrenF<>N3iZf`ngPm-DVB`A%70c9 z&}@LC@fu4I;BOvB8#JrC0j+ABV9q&s{a`sl(XBaIZ7Q7Cq0S0#A5p*wX!a^DY{*F+ z%)8<+aYi?QqV<>_>}gympmFA9=>rbW%RgvEXe+pQXp^O&_SA*~K% zRIDt^7$E5O9A1U0(`$_?1gi_^z7=(SDRrL~g2~XA?()+!0?E-+bVQ(|B*l|z1d%%? z=qofw3X@clY>ARr7%s}Y50NcgBp)Ln8MbQ|{GJ|zM-$j^4*5}EB|?`rd5hs-7Z26T zGiVMD$Y3=>vK^bzK5d+Mm&iqSoTn1W)U;b8X*zaGM1;qNX(groos-No-8nT*Cla0B z*~lctO?x>8%g%RF$yutCr@71V2+en@TJRn0tYq$v(MfuS<&H{Q<9z;%baW?}2$kTE zcI5Zr3?LqaLSKL$qJ7UVJU1_0n_ZZ__Qw1hzx4cv*TL2g>bDa9^r06RF3FHof(5&! zrfDu7A+ewo*q4*-mmYjE?les$<#85hZ6!&P!>3ZR z7TKBG0xMIrP}-_*we)wqI$puJ(tn_D7D&l?vC5L)Q$=wUeci?M!Ew3x!!41XC8jEXn^dLiqT z4%H_G88@aBT0D3K?zP63@U!K$*haDxZaCn8Q&`DhP2ZrS({CvF3oO~J67X&Eo75fF zQbc8=ST8SF#BWix2g(X8E{a%dAvXP%=VP<)*GgnQ@Rz+P5?Q<7{lw~pH=Ph6gQXnVIzuFqaXS01-}l4fu>3lc{jqS_S-hH&S-M4k`HnbtO|h& za*$>DPL9s_TWf->KTY0r{;Vq8;a6Ww6#kNW%EeEs}L3Kh})?wqUPb z&0zgYVbvD&E7WePymDS5i_a+JwbJ1Q+m{Xlih{T55o<1z8s91+cHjk4Ug8B@{w!wZ+>&mOZliT@)F!irCi*b)4N9miW8tHB3X`R zp_&rnOp;_K^3^I&PB0uUQ+X!EK>?U{4>Rf`BCB6e%B`BDHe7bSu zVw`0$7kZi%W)*yr5=dyZ=ouiN-qJ@4kx2|9Ga}EoL_1reV~5D?E&O(W+Hw|3&|Sx& z-W1Yh$Dt%8lk^py9A^Z#R&qeQ##{+=bTrQ4npH9i@oAb#36c@lt56@NkND^a9ls2+ zp7Q&>Q0eh3ljLcv8CDQ8A$LTWqyP*^DC}awB$VV)a(q^xd3U;fB&WU%$I(KHgPM&@ zGRa-9?A{FpBqgMHGQvb@jFfKi(U(#*FOXP_C<%h(m?)ZIUI519R+65HgyYExW;nfF zkxNFLpyCO@#G;j8rQK(7$t|}PG{*tNM2YEC2I>7?ih~k1209OUKs;Hy367bT6~W}y zf}V^}$!I)Efx79A?j1dcsH;3nCEdaC^P!_-ZAXKA+m7?0?yYWje2HO?PiEW#c zv;`cbT0E!I-jP%UGl4!NF{>H_oDFKI;=-xJC`GPM5BJ5BaefluaOOhAX7>@9DQ5`o*J)E9PW=KGo5R&KphU>wvvgH z!_mF+xD&@GjO6as+|#?;MuU=ClUQbpeR(R}MGneyn^bgjA0 z4VpJ~$W7cP*@Z>Pk~;>_SSS7&Bz0`Ah z`LK6gOO*MDCk|KE(Y)knzRhPGZL7}e8}{pVv0-R2ojuPioo6ymc4bS;t+98;=7-m- zk=Y?PqOhMaomcLW7Fx5Qre@K! zb+u{doq@Ylw&@u3x%1yysoQp|=bfJAy7qa`%qJA%dKZb zXY*=Z^KH*^UEf=#6=!YMNiI3b+cgiK+c-1qiM;}ly$T%Ioj5d z2F#-hcU$Hy{QtU|33qt$BX`~|G>-Qp zcz;a~jGnDKk+b%BxWUD9A|1zYQ$=vCsB+6XBCvf_iwDKflsGfHhM-gK*Sux>7ebo` zH1H7sp3l}GQU)~t<)`5gd|)_XA8={@u(N&0s#)B=d+3PnV^_nFN&oTwn!zglCsicO zKiN;<=Ybkewf=s!4d(Zo@yh*eHA7nc1FatB53JUqL;45RHADOK5BBNt{D^gUm+@0m z?eGrcr#tj8=S;xS?U0Q4s|`+n8jhAz(GJ*vw4i+Y>QIk+u13nGD9H&Qgi=s&bf~9Z zPFiM2lOFjSkSQEaQqy!eET>z?$=<%>XFdL~zkjIr^jSZLTU$Bj9fR4sh{H9qoFz=% z(@Zp#ppSC>u!g_+@jdVWLlMMh$o@I1{tP)jNBcfUElS+XeGvgIiLlbZf-HGf>Ba_S^aws>)tL`JG8p4ml1|271Y%hwd`;n9G16>%qWwkS8q?7y}E?9yVaW(k8%w^|J3#B1Id< zl4t8Z@_m2b$M^ld?`TJ`9l`jQ`uUwXKSE#O7kiM+hnEZRu#Rq{Byk1ljg`=eTPQ)< ztgUqz-6CT|<6Wnco(R(EBnh)ONx|HfBt}t`{wHR*mTmG`+}ICn@#UfiTT5G;cuLrQ zI6DX>Q3}O?504=fU{zE85mdKJP>BWDs zrq&@eidb|MO*Yl#Azp(>^bDaR5HtLj^p2vl;3LZN#fiuYoKB&>M~Q^kJY$qG zj$P1op{VGhlxKCBEyx_tYPzi4msMV4bE;fqb7Da_!%mKClO~}W{*s$LgsiSjbJ7|&tqJNq z5kAheB&$WPAQ~LDcS2b+gLy&EIC|ASP>P{J`!`JMXqRe#5dSz{?mGWRs!Gkj@O080 zyTQ<}$F|5vcWc3ue+!=cLfU1zA1-Vzl>27B(5uY*_3%h7Jh>B|tc0VraBL?WtAtN& z%HOpG=-mMNIcITyEsYy zX?z6cFMR}_{e2fl$S+4|c+NJ_8FV4!gC9nvm;V67e?7Ll(cH#OK)@Ep6ZWnP0a4-r zEVpr5@ti;kneud@OTR>bCMoJ5R8L#A5IdY;@Vs}}HWLm$x9E-?Iz`)f%lBjCTDpE3 z&ddWY4gyLiLM<8lDMi_uhfw@F!V*B@;R+7A5%{25K`bq{&^|7s*YSJ6mQyw34-oRB zw_Ybhy5jCs9@k#0Mv<AB?iPu&P_Ihm%x`0?4TB6n78Mh+77@ za}p5Lm*TEDwwsK&yrct00_VMM){?=f-)~dOwCOyEoY5=P^3^RebYQe0kIk zm`vKjLRhdV(~B1=Zw6fU0LtY10H+Oh`dCrCP!93j(qr|@Dj z*@D19?P+X!fq%k6KlLyc?NV{T1s9S{z&4a~r2PS=29hm&yig8(SiO0xe0#ZaGg}tK z>RqWUuU77^RVm{+)AjJ`=G9LY*5~WJk@ag|b@gsg&m#lBHGX4!Hc=ai?To}Kk(pZL z>`vrtC30?q`FkL;wNM>AU1$1h%Vdv8)%Ow(fcC%&1_TRMPQDSW z>}l&z!hsX4w=wKA{{K*t#&Bkhgaa7%4uA`pA^~Q4Etv5noEEUfk)o1>ora1h9(kiQ z^gs*gq;9JjTpG4_724Gz9RwES65Rre6m2Ca^ckFLwL^OT#fMT}WrUYKwpCa83E#ruu0Cm&w(+rqg zNpkvetV(^q-r4ie*fc)*X{~d3r*pW{Il6v%xBGCddu*qBY(4(0YhZ(__YBl}Mt6Ef zA6@)>q|$TzSH5pZ6dL>*8VBoQU)i3mQj5Ex;YU-I(5dySxTCNL?>!DzsjKxsZ<*>< zap{Qq%r@L?L13)_n0{zY78>1`FO!kDqL9fn!iBpODHZ zyFh~UK-$|dy&wsKc!oNjq2P0WSIyu5SAYN3Y{fse?)}yqB09dE3K9LUru@VqOV|GZ DbruzG literal 0 HcmV?d00001 diff --git a/services/__pycache__/similarity.cpython-314.pyc b/services/__pycache__/similarity.cpython-314.pyc new file mode 100644 index 0000000000000000000000000000000000000000..260ad844419c66885ae75b4f0ca0cf30b5a1a2d9 GIT binary patch literal 5047 zcmcIoU2GKB6~6PczxLW*{1+@M9)tfjHC~JjArvU73E-GiW=(fPu-VS;j_ozGJNBJf zV{cP+TO!3tg|ca&P?bnV5K*dB%0pH4p%1Ov$3ASkYn4uoiquLKQXfF3O6XJ1y|Xhr z#)gncxzgOZ_s;z}_ndRT^WFJcjn9i9vFgufe(gr+pQK_Yu3T7~he8gWK_O-s9b+`6 zjg;A8R%7AK4RgnMji+t=u;Z9hbGD%gcC^Nvrt)q_6YfFAsB4~(qZx&q%}B1(ydhV} z9b`ftXz`L3nboR7>~<8Wx<|U-VszJ#l1Xb4{#Jm|d36qJ@OF<$a~UZem*Qm7^2Epq zWg`08G-NpxMrYW)Zy}U`wpMh4nPx)FID9eEIBS-ATg(!I&3$eUt}-&nfuQ_7GLz1T zIup~KSkf{$9bhpLMt4lgibw=@PAWJhYAO+RoBm>OT1iR4nV2NY!O_L|w1`uqSecRH zni>?-i(*if@O&~ZsX;ZFO3ETmYKwidi+Xi5DyGwlCTdbNifdsq4?JoBvK;ypH8D*O zNOtwP8=W7v-6zG5U;mZ#%P1}Wv5|gUgU%>!?H(#Fc&OfQM!I{gJ%Y?`o0c~l?}&7V-GjWb%Qh5t zgSR`g!$V4HHlroeQ$k!(lW9q)_^1$*v~!Y_7SfrNgp+Ym7UsdJ6|4%%q##S_DQ#Nl zObB8+0k7;xn0^!z$&{2^k@#%f#aYDD)ZI0{dOaz4X1J;+=x-Z!dV=`4p zWmHX=7Uv~Fn+D;;lvG*Z-=@j36K_aZ5fV~bNhQ-Fh$sM_7)T=}gk-wBxu|V(J!v?^ zQ8B8{9q63w6=*GH)B=4%h@O0+mYsb&v#$<`ay%o$K~0zfZ_w60A)#n=Dw`OzURe~X z0hjKSm8mHS>yCsJ%S`DUXi|4b7%N!!hDA9ey+X@eHaUxlhXy>DxH~2guXC`9?y+Pq z7m46Hv!FAJdX;T?-CJ2(C3AFj2}Z&M*LA7(c{!gF@|Q_$cLSfkwDW-0u5vr4_~dnG<=^FZ*<&HKMK&|Z#(GgPo1<6)shu& zNC31{^9#ULkrh3n_w%Dpgc)NiSQKKKQJ5WILM)I5w_G>Ng}HW!_$E@)niYvMtm09; zW1enUoTLP}YbzS_n)RvPXFyY~5_85p<~p`Eui0k9Trk6!XLOSY27ph~nD=R4drf8o zeIY)PLq=2B6ZQ_Ught8kFkgqv*-Qy$9hF#C8AT`;NMpqqb2em+-#C*sqjS?4T`;5f zOU~#DJ3~B)WmDh}tq39;tvyp@b4RAuYdI1QyX+D33&lClIAKyWg?V$wk#J;TuzaA( zvCo?wsl+}_ISX-a1Evc=^aOIYqT+#0l?FOmi;-H72^CNYtV0U|+V$|Pk&yhQsjGzr z6^d#nw?V?IIf@E8yPz6Sz~Sw%B<>&?8UF+f(6^O=pX^5M+b=5++|`49c0$?-!__Xx z2t+WB!^{JkUmU#3U3%$0E#GMVxc;M7qul@on~Ni|KGi}W6!tR!3f9*!1BV?^hesRT z6!yhXeZn}J2%Q;#XAS%dX4;0@K&X@GNE;j^021@yC;Zg)m{DsTI+|wYp?Zxu$t<9p zqYaI;8#FD1wxiu#lDWmeorI!lfYaH&owyYS(z;64?-FuG%=GM^lG4(`EIz0QH`gL{ zCYx5R0YsbX^N@XZ4>nL}8u~r^oBo>%`MJy4_p=w@{NzNbdCwo%M=F8T^VKgf`IZ+L zOcsY5NDsUWX$YC^DhuN@*3w&_CJD7<1C06%;(2XTiL6^}8YpNg(tt{7z~+GN^iUc_{6EnP2Lvu}QSkl*P^6&^d@-pKwuHC9QtRwTJb~IT1cm*q zWk1j`v0s?6`}G*NzHSGCvgJU9gkYPN!vOf`7Q{ik0}48$>1NU7vVF=RV)j{RB^idj_)h2_QY^ zhg}g|D5vsH0ID|Z1QS)=V#}MESE;=PBY+d>@QaWI;aMX#rJmH-f9>=#bEW;V^keDz z=|^j1-#=_@{)%M7FTwPii$y2}{yNF_LzW$)w49Jtn?o7g)keArD}J&&Wuqr4I9Hgg z0ANMgINUMzflsDVN;+yB1c2gRi-&Kvkg(&R?S^xpvu|)R;Kl@|FwrJ%giPnpN{cGc zsZ5IUZ{0ahSh-5EHBj|92P1pdee1-)aG^uG(-5Hij$z_e7%s#J?}1Du+^wrCVes)o zhYWP7w!0@j3|%Nya|5LP&&I~j6uu)CCetNe`IINvY~b+&C#q@4zx<<9KXqQ2` zwN zE6(NZSACW!ZxqI3CHGl_n=dzgSi9oyvGH@f>Tk@aivEt}_$PZFa*X3BbGJg&1J!y} zT#;n}DewZ720PtFE>>`7(|NddP8+tXwe{)~V{zkGj%P~JjvKcycQl%YTP6H~p z%BkMzF{J2dSjN3Sr6e-4bP$g~7ZPGrauDxx48z<(ZGT5KcTnRU)N$8S_g7ELUpy^W b_7*)IIp=?!@Z;wGE+EyvYp8H-I`RJj+x+`+ literal 0 HcmV?d00001 diff --git a/services/__pycache__/sources.cpython-314.pyc b/services/__pycache__/sources.cpython-314.pyc new file mode 100644 index 0000000000000000000000000000000000000000..1752280c1bae6d67afc38187d40d92925ced319c GIT binary patch literal 2162 zcma)7O>7fK6rT02?cMk%0fHgFUdRu1Q*4P^0`b!d5rQH|soA8CXbslJJGK|uyVmTw ziA@jbsc?ey64VpN_QnyZJ#e5^j&75NXq!W&_R?D_P?5@|Z`QlEBSmUw<(W6{=X>9K zGv1vFhk6i<-^n+3hrO`UjXPLJcy$?sHFO6_%q%j|ZIt%7E}iuhy6s6at-Z-gtmHX? z%8bcL%ovJse^SGIzB`Tw4bAXMgn12RsfRt%ZeCXbtAv)Aq@#}lkE&+~Wr1&NWRT>6 zjNEz0;}Vl&YMvUKLrKS@%68%V-MG#wc{3>Ku#2J$n-)4YS88K%|AFPp2Hcno=k^5f z+x#k5h$!3dvhg8hh<*; zT_{n?rRwiG&BM5aZ~1L5Tq}}tfgA4s(62J8Q)8^RapkIM;!4$2wW4V1Vp*3J(a>uo zj}5Uv^om$e%lM*rX2LiluBc{7RB&0Xs3s;C#R)QRF{I(GnT4|#VvNOOEw3wp)QBzA z@_=9zYUOfW)T>y-ijC9)D(FNcShihHOCX37inv@iv60A^WFqGQ-bqgrH_jh`qmiPl z0nch#HEC#zu!uF2s4j0i@Z5r3iYS|+swsGtI;)0A3kF(5Tq!{%Mpe$^v$|HU(~=s! zPRyqh%56{>jWI_-J0WViDHik^WCzTuY?fNRS+vVv&F56Dpf}E?F{!AAq3Rm+OH-lv zl@iuO*=`DLwZpj6+{XBf4EPE?1iQeNM^_gpmsr^HmGxo~6N{snEROcp@+x?_R#KPT zS47uhyv5O?Euno4EWg9l3bceAEz0uebsFwM&789*qk!jwpK%qw5o>6V z>-jSEMXK5NQS;`<+nl@`KDu^e-_!36?}d&$8h^|_DZUKFe+k8(PVEJcKAd|n*BqFB zu51TWyOC2bBa=TxCbuH*+`rK_Uisd5e*U}9Hy5{q(r#pAqq1IkGP@Otw_H*WQq7Y$ zw-++ar8`>-xu%S_gN41}i6_OK;rKfLh}}IgxKUrPZ`}Lp-d_LU#{Bwxb8M-3E4$Ob z{0Hpk*5{h1&ON=h(|_eJ0gWWygituK|1Jt1Tbq9~1@cZP?rh(_K8}JTuaO5z;lmGu z4}#6(7oSaUChrHg`SdOy+6xRGw9alOo1fgF=A3QrKWM)6Y;n`1)!+|(G!>+5?&nVs652ddGx)R5c@;mG@auNof88yJo_W>1sVPc zOi#>f`RPTaH%)*On`>QKK9|$v3eM#SeTUn}_bqw_A|fL&S^VXSuGGr-12O?y`rS9E kexGF+=2sN{P3U_m4E`((KEALejIa6r_6;z{$T)cX3zntl9RL6T literal 0 HcmV?d00001 diff --git a/services/headlines.py b/services/headlines.py new file mode 100644 index 0000000..01614d4 --- /dev/null +++ b/services/headlines.py @@ -0,0 +1,94 @@ +import logging +from re import findall +import requests +from services.normalization import normalize_headline +from structs.headline import Headline + +logger = logging.getLogger(__name__) + +DEFAULT_TIMEOUT = 10 +MIN_HEADLINE_WORDS = 3 + + +def is_headline(text, stopwords=None): + if not text or not isinstance(text, str): + return False + cleaned_text = text.strip() + if not cleaned_text: + return False + words = cleaned_text.split() + if len(words) < MIN_HEADLINE_WORDS: + logger.debug("Text rejected as headline (fewer than %d words): %r", MIN_HEADLINE_WORDS, cleaned_text) + return False + if not any(c.isalnum() for c in cleaned_text): + logger.debug("Text rejected as headline (no alphanumeric characters): %r", cleaned_text) + return False + if stopwords is not None: + normalized = normalize_headline(cleaned_text, stopwords) + if not normalized: + logger.debug("Text rejected as headline (no meaningful tokens after stopword removal): %r", cleaned_text) + return False + return True + + +def prepare_headlines(sources, stopwords, timeout=DEFAULT_TIMEOUT): + logger.info("Starting preparation of headlines for %d sources", len(sources) if sources else 0) + headlines = [] + if not sources: + logger.warning("No sources provided to prepare_headlines.") + return headlines + + for idx, source in enumerate(sources, start=1): + if not source or not source.strip(): + logger.warning("Skipping empty source at index %d", idx) + continue + + source_url = source.strip() + logger.info("Fetching source [%d/%d]: '%s'", idx, len(sources), source_url) + try: + response = requests.get(source_url, allow_redirects=True, timeout=timeout, headers={'User-Agent': 'Anya news bot'}) + logger.debug("Received HTTP response %d for '%s' (content length: %d bytes)", + response.status_code, source_url, len(response.content)) + if response.status_code != 200: + logger.warning("Source '%s' returned non-200 status code: %d", source_url, response.status_code) + source_content = response.text + except requests.exceptions.Timeout as e: + logger.error("Request timed out for source '%s': %s", source_url, e, exc_info=True) + continue + except requests.exceptions.RequestException as e: + logger.error("HTTP request failed for source '%s': %s", source_url, e, exc_info=True) + continue + except Exception as e: + logger.error("Unexpected error fetching source '%s': %s", source_url, e, exc_info=True) + continue + + logger.debug("Parsing HTML content from '%s' for headline candidates", source_url) + try: + link_texts = findall(r'<(?:a|span)\b[^>]*>\s*([^<]*?)\s*', source_content) + logger.info("Found %d candidate tags in source '%s'", len(link_texts), source_url) + except Exception as e: + logger.error("Regex extraction failed on content from '%s': %s", source_url, e, exc_info=True) + continue + + source_headlines_count = 0 + for tag_idx, link_text in enumerate(link_texts, start=1): + cleaned_text = link_text.strip() + if not cleaned_text: + logger.debug("Skipping empty tag text at position %d from '%s'", tag_idx, source_url) + continue + if not is_headline(cleaned_text, stopwords): + logger.debug("Skipping non-headline tag text at position %d from '%s': %r", tag_idx, source_url, cleaned_text) + continue + logger.debug("Processing tag [%d/%d] from '%s': %r", tag_idx, len(link_texts), source_url, cleaned_text) + try: + normalized_headline = normalize_headline(cleaned_text, stopwords) + headline = Headline(cleaned_text, normalized_headline) + headlines.append(headline) + source_headlines_count += 1 + except Exception as e: + logger.error("Failed to normalize/create headline for text %r from '%s': %s", cleaned_text, source_url, e, exc_info=True) + + logger.info("Successfully extracted %d headlines from source '%s'", source_headlines_count, source_url) + + logger.info("Finished preparing headlines. Total headlines collected across all sources: %d", len(headlines)) + return headlines diff --git a/services/normalization.py b/services/normalization.py new file mode 100644 index 0000000..be83364 --- /dev/null +++ b/services/normalization.py @@ -0,0 +1,56 @@ +import logging +import string + +logger = logging.getLogger(__name__) + + +def get_stopwords(path): + logger.info("Attempting to load stopwords from file: '%s'", path) + try: + with open(path, 'r', encoding='utf-8-sig') as stopwords_file: + logger.debug("Opened stopwords file '%s'", path) + lines = stopwords_file.read().splitlines() + stopwords = set(lines) + logger.info("Successfully loaded %d unique stopwords from '%s' (%d total lines)", len(stopwords), path, len(lines)) + return stopwords + except FileNotFoundError: + logger.error("Stopwords file not found at path: '%s'", path, exc_info=True) + raise + except PermissionError: + logger.error("Permission denied accessing stopwords file: '%s'", path, exc_info=True) + raise + except Exception as e: + logger.error("Failed to load stopwords from '%s': %s", path, e, exc_info=True) + raise + + +def remove_stopwords(text, stopwords): + logger.debug("Removing stopwords from text (%d chars): %r (available stopwords: %d)", len(text), text, len(stopwords)) + # split() without arguments handles all whitespace (spaces, tabs, newlines) + words = text.split() + sentence_words = [] + + for word in words: + # Strip surrounding punctuation and lowercase for comparison + cleaned_word = word.strip(string.punctuation).lower() + if cleaned_word and cleaned_word not in stopwords: + sentence_words.append(word) + elif cleaned_word in stopwords: + logger.debug("Word filtered out as stopword: %r (original: %r)", cleaned_word, word) + else: + logger.debug("Word dropped (empty after stripping punctuation): %r", word) + + logger.debug("Filtered words: input=%d words, output=%d words -> %s", len(words), len(sentence_words), sentence_words) + return sentence_words + + +def normalize_headline(text, stopwords): + logger.debug("Starting normalization of headline: %r", text) + headline = text.strip().lower() + punctuation = string.punctuation + for char in punctuation: + headline = headline.replace(char, '') + logger.debug("Headline after lowercasing and punctuation removal: %r", headline) + normalized = remove_stopwords(headline, stopwords) + logger.debug("Completed normalization for %r -> %s", text, normalized) + return normalized \ No newline at end of file diff --git a/services/similarity.py b/services/similarity.py new file mode 100644 index 0000000..7fae8f0 --- /dev/null +++ b/services/similarity.py @@ -0,0 +1,57 @@ +from collections import Counter +import logging +from math import sqrt +from collections.abc import Sequence + +logger = logging.getLogger(__name__) + + +def cosine_similarity(a: Sequence[float], b: Sequence[float]) -> float: + logger.debug("Computing cosine similarity between numerical vectors of length %d and %d", len(a), len(b)) + if len(a) != len(b): + logger.error("Vector dimension mismatch: vector 'a' length (%d) != vector 'b' length (%d)", len(a), len(b)) + raise ValueError("Vectors must have the same dimension") + + dot = 0.0 + norm_a_sq = 0.0 + norm_b_sq = 0.0 + + for x, y in zip(a, b): + dot += x * y + norm_a_sq += x * x + norm_b_sq += y * y + + denominator = sqrt(norm_a_sq * norm_b_sq) + if denominator == 0.0: + logger.debug("Zero denominator encountered in cosine_similarity (norm_a_sq=%f, norm_b_sq=%f). Returning 0.0", norm_a_sq, norm_b_sq) + return 0.0 + + similarity = dot / denominator + logger.debug("Calculated vector cosine similarity: dot=%f, denominator=%f, similarity=%f", dot, denominator, similarity) + return similarity + + +def cosine_lists(a: list[str], b: list[str], *, casefold: bool = True) -> float: + logger.debug("Computing token cosine similarity for list_a=%s and list_b=%s (casefold=%s)", a, b, casefold) + + def tokens(xs: list[str]) -> Counter[str]: + return Counter(x.casefold() if casefold else x for x in xs) + + ca, cb = tokens(a), tokens(b) + if not ca or not cb: + logger.debug("Empty token set detected (count_a=%d, count_b=%d). Cosine similarity is 0.0", len(ca), len(cb)) + return 0.0 + + common_tokens = ca.keys() & cb.keys() + dot = sum(ca[t] * cb[t] for t in common_tokens) + norm_a = sqrt(sum(v * v for v in ca.values())) + norm_b = sqrt(sum(v * v for v in cb.values())) + + if norm_a == 0.0 or norm_b == 0.0: + logger.debug("Zero norm detected (norm_a=%f, norm_b=%f). Cosine similarity is 0.0", norm_a, norm_b) + return 0.0 + + similarity = dot / (norm_a * norm_b) + logger.debug("Token similarity calculation: common_tokens=%s, dot=%f, norm_a=%f, norm_b=%f -> similarity=%.4f", + list(common_tokens), dot, norm_a, norm_b, similarity) + return similarity \ No newline at end of file diff --git a/services/sources.py b/services/sources.py new file mode 100644 index 0000000..d652326 --- /dev/null +++ b/services/sources.py @@ -0,0 +1,31 @@ +import logging + +logger = logging.getLogger(__name__) + + +def get_sources(path, delimiter='\n'): + logger.info("Attempting to load sources from file: '%s' with delimiter: %r", path, delimiter) + sources = None + try: + with open(path, 'r', encoding='utf-8') as source_file: + logger.debug("Successfully opened file '%s' for reading", path) + content = source_file.read() + logger.debug("Read %d bytes/characters from '%s'", len(content), path) + sources = content.split(delimiter) + logger.info("Successfully read and split %d source entries from '%s'", len(sources), path) + for idx, src in enumerate(sources): + if not src.strip(): + logger.warning("Source at index %d is empty or whitespace-only: %r", idx, src) + else: + logger.debug("Source [%d]: %s", idx, src) + except FileNotFoundError: + logger.error("Source file not found at path: '%s'", path, exc_info=True) + raise + except PermissionError: + logger.error("Permission denied when accessing source file: '%s'", path, exc_info=True) + raise + except Exception as e: + logger.error("Failed to read sources from '%s': %s", path, e, exc_info=True) + raise + + return sources \ No newline at end of file diff --git a/structs/__pycache__/headline.cpython-314.pyc b/structs/__pycache__/headline.cpython-314.pyc new file mode 100644 index 0000000000000000000000000000000000000000..24fe8cb7f651427311605405069fa651be29dd1b GIT binary patch literal 2815 zcmbsrO-vM5_`R9^)rAFg)#XP95z1P!#DGPs;t!Z=C1g7y6a}Wk?65n!%xvF0L}=P< z+9uVLrzSS(p}lFsfkO}Np^2?Mcw}9YrBj;Nv==W8im9Z%^m{WqfRx52eaU?D-uHd) zz3=~fua=i>Loljm?p*oZi_ouZ;SNt}vLeDHhenZz&mqeG+&O%fqg*{2K?$GBzW#4O zBc2ZItf@DG8j#5EMxv*KE8%=1ZbUKfW47#zVJoBMl-3YI7|*K?LaF%u$DoJ z5k#?wC^zb-{4Qjm0o0B~yc=Z#A_uU7de~jmYvVl(Pko}7`UemVh(5U0Zw1b(axw$= zDNM$aoCvf>JbWv!0XT+paPla3_+x~UfUQMu;WXTH0bd5~0ot?HYG5smHm(ufqUv>Og>{2W$Qf-$P1=aNoW+}zSRSTrjI(^>j83gGNm4ShX(ly=N|MP= z+LyCB+S0~^+ID43&1BjV(@I(<6A5BmQ59;o$@;Y124_vOMY^O_JUeZLBuUnFgUVEu zBrD=79N&oE4i}h>09JG8Ir4ai@Dm$;d12Z9f79_D|LqY6w*qWPVdim%A11I4tb7iW z92%_9T?!f$bl0e8*yExTY{#Y{f{_!OaBMjobdf-JL%>DB-wm#Rp_cBfhkyl#b2HuN zbd73`3k25|f~K2P))iG~+32q0O{7(z##L396k}4S$D5L|*M%mtFXkmoD#G%`@@0%v zN+p&zsg6yiEIy;^G2Zf+YG!<$%VhRyZf+(OY)e3rG!O&aQR#{R?r>1HD~5-moW^O5FR{!UM^^<=*FWTEBMy^dn{V7`0s4;}}H zFxdeIC>yO60)*|OHvyYN!)sm{z{6{NZc@Gi%xt6}{%ik>-vSQ*MmWqAvBv;*eUxBQ zY#Kma^U;O~32$P9rHH}DTisOY@(y5;8)mB8WBuNaQnb4NdQ^jkBWrvpvCEyW{|E3^ zSl2>X?j^GFgTnumbXXaicF*Zf(<`1vwRg zFy#r070W**6CKpF1+qL$36{yaJOPE3un@BRDHZB0E9gYc62ijHDH6dm*A+EOHAA<& zDibT_A0GZi{<-pdRz#2^9WnUmzwX{?CVrQ}8P`=^N+y}+Zi}}urg`ER8 z&anG>Z}!eddl#d7i^9RYaPX_nue%;}EnJr7-_Z+~jY4~tT}kkx9eZx{t}cc5JVQJj z=x*#@jMf#S2lCMavv17x-RoQEe|vr?Rp?I_TDAGBnfa;rmU&bbeTHz^`}k3)@~>q- zZ2!}Yg7V*Ur*?Hm(f#Q5o_+lN{YQH0_#f)LFn^)hGdnX!t(*q1b+d0ntl1ysMBEHK zTeWDYwghJ-<^~*PCQy+NgCIPN!jB@ zHy}WnMMw+VRJoz`k5B3fdltYe10q*bBo@Bf*d$d=)}A)JLlk&*NKyR5CH>>P{wCAAftgHh(Vb_lhJP_LlF~@{~08C%Sb;XKQ~oB zD=9T6M?au4IU}(sH=rm#D>b>KSU)kZGEu)IwHU~ZkI&4@EQycTE2zB1VUwGmQks)$ XSHuc50%S=si1CS;k&&^88OQU9 zkhI=ZUUDe@K9~Po=Rg1DJH31AVg%BIKYnm?qm_^}7W@?UGFy)!vrJ})%v~fF=Y07K zLZ0s=7eZFJ(`|`Z5oqBrMlXn#c%jCs@#T|^G; zCvx~kWa5xNW$8OCxOuC=;fw%h^c9?qH=1Pe)Q5y*nhEJ3z&6qj9nPs%dcm$4Q|I;cn3=wn5bVYr)p9D;lm$)AWOYMhcCF5Wl3jo9{qv*O zKA2P{FOI!`<=W(qL-e*mw*D4oP~jMFz!A#C;%B0k&_YaDzWiA0x6o@4f@fdz9>8@M^EWwQelOX&rbE>2MMrk1wYfNCtM0~RcjIW#zY zmdq{LVWwHva`u60n@BosBFw;QfGiWc?c_@6ulH<0foyj&jo#7I8cPPhZIWuY^Egp* zUbE7;Z9hdJ;~c>dEO?nr2glTjU!V&w`FQAQ`0Wz{+EoM_Avu0)~HKIxj zRGWME_~GNJbK0Dm&svfvfs!whBLiw`WWZ$~aQYU`{O)6Hw=P&#j=eoFpna?^=CWEc zZ7vQN+Fdre9lHji(LXw`8P;~)C`g}#_0p_qJrwMEHJdf>Dpbqp6vfkyTJDvxqbNrj zWrfz=wlbcGQl@Ov6EG=->&>9 z?d)?P%jDl$yFRO3soOkqe5Gz(?Anwx{2tsqsN(m)rqmCmCwu#gP05w$EzbVEy z#MV`@_4B>=KKks?gT{5SH>kVUc7O2iIyTydR@;V(!{c9zS18JLbtUlc&eH#{mB36) z4xRLr$V(JK82BP`6cm8CLjg3xqASvLpcJYr0Qc(BHt5Dif$Vne&gMa}FTT#r9sQk$ zps0Eta1F9s~?kP)}q8Pr!uPw!wx43&l{z6>z{y(aLZtmuL_RG`upgyP6uZ( z;o1720x}b*iYt;#;1sLuR_6SZ>BBj_zE#5AyJ8RlG4_BKqmFk*Haazmaios@kPKGUjDXgY45+Pg2)2to2byubH z%&xvGWEkzUw1sE7!fMb2qvTGu1nV!)UZEg;Z#-7<-Q~1V{-O_VRaR&3&si;Oq z$x#zbG1Hi0f=f+VawKbl*=MO_+4%i1Gy4e0ZEwImqWMLT4=F>KJGaB+2xh7&7R(svGw@T%_Ay)yEeOS z;P=qxUa1(D9Hl`AVT$d&RerAVM3knWu}rfbgQ6Wn4+UJCykWTtwg+l~7ab7#26Bkb zX~OZj{rj*1w?p3r!cf*6w?jJZp6alpjDO2BvoHRy9Y3z0Edha(ezJFXMchR4pP^_? z*V85<^{$H@PxiLo`>@z_tjtGwbbGz|B5H2W>$umt(bl)x*7rbN7vBv0j#v947rz!S zy{NwNvO4P94M=2?<3V`~-$9K8k>@M=1}Xq%_L404Y9+}}aiyT<`eDaVl!BVmFAZwY zNK^@GK)p5iGnKv-Mxsi;GOwm&BuJTh4}l5`7utRG4yp=OS-&S>7zh-C=m4}-W=BiC zr}-SU2THFHmx0AEp$7u`oY0d%>^&Z@uWf3P)hkrHIuxBO=guS|W!Gh|x5APm0-C7- zS5{s_gKNZ4ce^*py+HlW1*#kcHO-DeP#fjIn?JMr&3WB1-h&mezW@R^=FmGuv3;}s z_ljcc^9ZTiw-N7LjdyOuldJJ$ap3e?eDra=*{|$fjrTqnUyHx%(->(! zUTk`+>~4Af|7+bW-}|8`DiDF$qM{h;qNXTzOi>oiOg@YGxT0X}G=X@E8q1#pkL zvP$R%-WPH897UJNA$Zl;9U#B?AspbhJQqIoEJ7l+#oE5LaQ`zt9-iQ~1QJO;b+XSo z%GzqfeO?pfo*gJ_X%1iZdtvrjUs+q7*X!3jnPeSl-bLT<%mGgW3ra+ya^y=T?6fV!_1tkFS z9Bjt_J-i(WHv!Ic<%B{;>u&Bp=88!Gd-0Yid6AA9X2mtl7`WF}P@isTi-qG7TQKw4 zjAWRWG^ICP>Q zB$LTPR5v8>U{!GUqAWj)zqMWKA}Qy=-_j_arvtwv1Q$T5Gfbm|K++j#XZS+^OycU= zqj2>trJPHFP2F-(7zX+lF-_31BfH4eYxSPQS<&@F<9krwVwL7li|1*)Mr{OH<4IqWXz6ifme8WzU z!v0-gk?rGGHAA&DDTk8Aq($|nNhRHwqYz`zd>YU{O`6rS*}~DQX&rz;9S~^=(7d#H z2f{1BA|(h9R6}Ap4cggu?|BVyM>Q+S8e-a-v|wg43`T@#41`glFd4*T7Tm9;M2IfH zPVAT~<&zXJ#utRV9Wb8V!?^08c^N zRFQn0okKSnZB#|7>h6~5^{H?JxfaRuNs{N!!}z!7CeWFwNaOrvaJEt3=c2g?%!J2O zkpKyZ`m1p)P&J-V1rLIU6-f@i22bQQc%rYtBfbVtO*xMoI}9h)3IDl>P7<~1KZ6;L zf!%PD3q}M*taG0-c2qgl-rGj^06arEmascjX&dov<7FMqw612mQ?4C+1+#UhJ5@52 zuVBua^SaTU@{H*#m{kfEHmeN+|LZBP>f<@!4q9yw=Xr7C4##hy%&HCKp+~{sBkr3~ zP%eb?)?EK72S%nQNG8o8L5G7XNWPCmMKTLyJW*G|hwO-E;FHbn&@uubCD6kbT`ddl zisw~~o7P1mA(jPqaeUGOw8VJ=qP^HZ2;^*T!LEnb8w%bGJo(^54rA!ny1||Aa=hp0 z;ul`3A%e+*1pp8!4-ScXN1ulX1-uuG`^x39LTYvX)F@jBOpZ;vfHDO&ak4@x&`1eqMaY`Ii$TYX{!_@~sEz-={uLefic(eeuA% zfU^8e-CxxeTi$!rxh{S$D4u0+;RWYb@61W z+;;!Ux;R`ax7_diS{(fEZ|g~W%89TflhM@ohbPI#EmF#VU8ie z*oOW9$pjJ%EvO8nJV3mLl}M0<4FU-Q#LL{bk>>En+_!zPFyo%Jmc@3R;UOm?{520b zH@3quFb-Z+bwg1S&9*q32Vv0x8jYXMEh#~%P01~xbR}Iu2JwM6^1BrLdAla4vZH=o z3_3f)s8U6~N0 zBeg@gfrrH1l=Da_j<^whiv9=~Fo0rz2;`~2aojhA{5kov`ftb^|4M+4uafcSk@Fna zckk3jd*5n%-*bYc=Q0nUp?e>0989hrOgfcAPi3Bj;wy!9p=~o1|KZr5j{W78X95Ww HaOVCWsQ5-K literal 0 HcmV?d00001 diff --git a/tests/__pycache__/test_headlines.cpython-314.pyc b/tests/__pycache__/test_headlines.cpython-314.pyc new file mode 100644 index 0000000000000000000000000000000000000000..c82135de3a67b47d5c147bb628f39948699c1da4 GIT binary patch literal 5146 zcmcgw-ER}w6~8m%k4fTeAPM;(fjET#ZX6rZw6KuP!tx=q+k|KcW~of3gFQ(M_RMha zH6eM}My=Y-Lsj*ms*q@mnDg zOI#W#_xrdH`lU(#^#BiC5AxsuIYb7D;_D`g|1Oz1;4NA$$E0$dH{ESk0_Q#^B;83! zKLPHpDwkIqBx7!M0Q5NK{35|Gv2Qve+I~ypsYSbKN?XXJrp(j>=sRdp-O{M8F6vr3 zn=y1|w_UkX57dpp?mew}ux`v>GbY!K1!~5clpt?T5Z-i{Qf;Xn~l#brxrDuo1x}?J;hmEGnTbD z2PK(9tEV5cmTf=N`7O&n+-MUqXG{hOtO>{}v3t+1`F{4!E;Pu>0MqHCOiE|5%5}!5 z{xFY%VhcJ?;kLam02R1Sc{d6Quaactc)aVd#JMjLqEYEr1ZY)OP()~X=0I)_8a{7f zMT!ZQlx*?Zw@=T6yn3}9b62u_31OzK+V9mAN%18sE}{6(y0QjHg3NSyT-7pp$g3xO zp5e_2o|OG&CBP-4hrOAkp<5||>MBYQ%tHS13z)&q0vTWAx$MOFJu|&LF|LIdsXqVS z$zvxc@b1E%1&+tHiSc-uC+S49twHDG;WoRnti6>QaU;QZy*Lr967|VP4DHQdT$YP&&h`thTJ; zs_TAw5ZUA$;ExI>X}3Bxsx!+p7%0KaJOvB3n<^&EQ1*!v;c%F?@sn$p>l#V2@JX{KoK7{&NOFmVUPZc7U3%BlUrf5MvwJEcb9NLn5igM4d zI-cBldi2@BO?kM|_oVl;Gr#WN>K!fij{fe(bNLEI`)%x?e|9wde>>=0v*Pod?hQ`r z2fl#E=eQYOaNU=zIBa9*EV~u$hH2~)ko~;lavtC|7j_Z-O%8VHzg~i3Xjm6$9~85+ z7wB}P*>M}%4-I+*$x$Q&NGc9O!`K`|@(m=%fz+FDk7JT4XE%E-cw~PIZorPn#U21@ zV8ChDfWQAH1OA+Dc7FJp2^UhjW5V+lbM;L4xjg&^Lv4pxp#@fnLd<|7SYjgZEkO~x zNmjO`+h?l5^=2R;#zf1Ptj3RvGYCj^-<+gKo@rTr!2Mty12iFeU^uM%=EN#yH=b9G zPYM4%@uHD5e_gz25ZM%8Rkq@XcoA3`@KBopK97_kE>?r!Rfc(u@dgZgk{1M8w94Jz z0~r@qAtK(CoaFF3m>ec6XEEc(*h&XfX3$XUJ_S#)!EP17JDDiTbU&|eQAlDA;(T>E zLq&#@B~<1r666dOISCQY#BbcYdkgSEZd9Ks&5DPOM!cv+!ZphAq zusx2=EX|nsfa=3o&c(okmT39BntNQ#r9F0A< zKPq_fk3Gm(Om!A>^;9(@ty{b+5f zbZU5QYBMxcI<8?kSQ@;G<rP6yw+FaywU=KiVC8Pk0*TQV$3@A%Sf!Oc3ydK)LAgkMGu}46@`V zX&D0)hjYst8IG1@w|cc0z5~n0;zQ-p^zHW7|IjFh2sBGuM&r7y0gtoYYuM*Q8SA7h z*Lg0=H)ptHb$XT&CeY)gV5{E@W%iG0(cmtF;~2}1PAxvjovu`l|HU~5|ui7&La YvMKeJe4$ULemM1$8#@y59d_3L7gR&PCIA2c literal 0 HcmV?d00001 diff --git a/tests/test_headlines.py b/tests/test_headlines.py new file mode 100644 index 0000000..3c13aa5 --- /dev/null +++ b/tests/test_headlines.py @@ -0,0 +1,154 @@ +import unittest +from unittest.mock import patch, MagicMock +import requests +from services.headlines import prepare_headlines, is_headline, DEFAULT_TIMEOUT + + +class TestHeadlinesTimeout(unittest.TestCase): + def setUp(self): + self.stopwords = {"the", "a", "an", "in", "on"} + + @patch("services.headlines.requests.get") + def test_default_timeout_used(self, mock_get): + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.content = b"Default Timeout Headline" + mock_response.text = "Default Timeout Headline" + mock_get.return_value = mock_response + + sources = ["https://example.com/news"] + headlines = prepare_headlines(sources, self.stopwords) + + mock_get.assert_called_once_with("https://example.com/news", allow_redirects=True, timeout=DEFAULT_TIMEOUT, headers={'User-Agent': 'Anya news bot'}) + self.assertEqual(len(headlines), 1) + self.assertEqual(headlines[0].display_text, "Default Timeout Headline") + + @patch("services.headlines.requests.get") + def test_custom_timeout_used(self, mock_get): + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.content = b"Custom Timeout Headline" + mock_response.text = "Custom Timeout Headline" + mock_get.return_value = mock_response + + sources = ["https://example.com/news"] + headlines = prepare_headlines(sources, self.stopwords, timeout=10) + + mock_get.assert_called_once_with("https://example.com/news", allow_redirects=True, timeout=10, headers={'User-Agent': 'Anya news bot'}) + self.assertEqual(len(headlines), 1) + + @patch("services.headlines.requests.get") + def test_timeout_skips_slow_request_and_processes_others(self, mock_get): + slow_url = "https://slow-source.example.com" + fast_url = "https://fast-source.example.com" + + def side_effect(url, **kwargs): + if url == slow_url: + raise requests.exceptions.Timeout("Connection timed out after %s seconds" % kwargs.get("timeout")) + fast_response = MagicMock() + fast_response.status_code = 200 + fast_response.content = b"Breaking News Story" + fast_response.text = "Breaking News Story" + return fast_response + + mock_get.side_effect = side_effect + + sources = [slow_url, fast_url] + headlines = prepare_headlines(sources, self.stopwords, timeout=3) + + self.assertEqual(mock_get.call_count, 2) + self.assertEqual(len(headlines), 1) + self.assertEqual(headlines[0].display_text, "Breaking News Story") + + @patch("services.headlines.requests.get") + def test_connect_timeout_and_read_timeout_skipped(self, mock_get): + connect_timeout_url = "https://connect-timeout.com" + read_timeout_url = "https://read-timeout.com" + + mock_get.side_effect = [ + requests.exceptions.ConnectTimeout("Connect timeout"), + requests.exceptions.ReadTimeout("Read timeout") + ] + + sources = [connect_timeout_url, read_timeout_url] + headlines = prepare_headlines(sources, self.stopwords) + + self.assertEqual(mock_get.call_count, 2) + self.assertEqual(len(headlines), 0) + + +class TestHeadlineSafeguards(unittest.TestCase): + def setUp(self): + self.stopwords = {"the", "a", "an", "in", "on", "and", "of", "to"} + + def test_non_headline_link_texts_rejected(self): + non_headlines = [ + "Account Settings", + "Follow", + "Television", + "Sign In", + "Home", + "Politics", + "About Us", + "Contact Us", + "Menu", + "Search", + "", + " ", + "123", + "...", + "in on a", # only stopwords + ] + for item in non_headlines: + with self.subTest(item=item): + self.assertFalse(is_headline(item, self.stopwords), f"{item!r} should not be recognized as a headline") + + def test_valid_headlines_accepted(self): + valid_headlines = [ + "Breaking News Story", + "Custom Timeout Headline", + "Senate passes major infrastructure bill", + "Scientists discover new ocean species", + "Federal Reserve holds interest rates steady", + ] + for item in valid_headlines: + with self.subTest(item=item): + self.assertTrue(is_headline(item, self.stopwords), f"{item!r} should be recognized as a headline") + + @patch("services.headlines.requests.get") + def test_prepare_headlines_filters_out_navigation_and_non_headlines(self, mock_get): + html_content = """ + + + Account Settings + Follow + Television + Sign In + Senate passes major infrastructure bill + Home + Federal Reserve holds interest rates steady + + + """ + mock_response = MagicMock() + mock_response.status_code = 200 + mock_response.content = html_content.encode("utf-8") + mock_response.text = html_content + mock_get.return_value = mock_response + + sources = ["https://example.com/news"] + headlines = prepare_headlines(sources, self.stopwords) + + self.assertEqual(len(headlines), 2) + extracted_texts = [h.display_text for h in headlines] + self.assertIn("Senate passes major infrastructure bill", extracted_texts) + self.assertIn("Federal Reserve holds interest rates steady", extracted_texts) + self.assertNotIn("Account Settings", extracted_texts) + self.assertNotIn("Follow", extracted_texts) + self.assertNotIn("Television", extracted_texts) + self.assertNotIn("Sign In", extracted_texts) + self.assertNotIn("Home", extracted_texts) + + +if __name__ == "__main__": + unittest.main()