| @@ -0,0 +1,10 @@ | |||||
| # Default ignored files | |||||
| /shelf/ | |||||
| /workspace.xml | |||||
| # Editor-based HTTP Client requests | |||||
| /httpRequests/ | |||||
| # Ignored default folder with query files | |||||
| /queries/ | |||||
| # Datasource local storage ignored files | |||||
| /dataSources/ | |||||
| /dataSources.local.xml | |||||
| @@ -0,0 +1,15 @@ | |||||
| <?xml version="1.0" encoding="UTF-8"?> | |||||
| <module type="PYTHON_MODULE" version="4"> | |||||
| <component name="NewModuleRootManager"> | |||||
| <content url="file://$MODULE_DIR$"> | |||||
| <sourceFolder url="file://$MODULE_DIR$" isTestSource="false" /> | |||||
| <excludeFolder url="file://$MODULE_DIR$/.venv" /> | |||||
| </content> | |||||
| <orderEntry type="jdk" jdkName="~/PycharmProjects/anya/.venv" jdkType="Python SDK" /> | |||||
| <orderEntry type="sourceFolder" forTests="false" /> | |||||
| </component> | |||||
| <component name="PackageRequirementsSettings" /> | |||||
| <component name="PyDocumentationSettings" /> | |||||
| <component name="ReSTService" /> | |||||
| <component name="TestRunnerService" /> | |||||
| </module> | |||||
| @@ -0,0 +1,6 @@ | |||||
| <component name="InspectionProjectProfileManager"> | |||||
| <settings> | |||||
| <option name="USE_PROJECT_PROFILE" value="false" /> | |||||
| <version value="1.0" /> | |||||
| </settings> | |||||
| </component> | |||||
| @@ -0,0 +1,4 @@ | |||||
| <?xml version="1.0" encoding="UTF-8"?> | |||||
| <project version="4"> | |||||
| <component name="ProjectRootManager" version="2" project-jdk-name="~/PycharmProjects/anya/.venv" project-jdk-type="Python SDK" /> | |||||
| </project> | |||||
| @@ -0,0 +1,8 @@ | |||||
| <?xml version="1.0" encoding="UTF-8"?> | |||||
| <project version="4"> | |||||
| <component name="ProjectModuleManager"> | |||||
| <modules> | |||||
| <module fileurl="file://$PROJECT_DIR$/.idea/anya.iml" filepath="$PROJECT_DIR$/.idea/anya.iml" /> | |||||
| </modules> | |||||
| </component> | |||||
| </project> | |||||
| @@ -0,0 +1,6 @@ | |||||
| <?xml version="1.0" encoding="UTF-8"?> | |||||
| <project version="4"> | |||||
| <component name="VcsDirectoryMappings"> | |||||
| <mapping directory="$PROJECT_DIR$" vcs="Git" /> | |||||
| </component> | |||||
| </project> | |||||
| @@ -0,0 +1,72 @@ | |||||
| import logging | |||||
| from services.headlines import prepare_headlines | |||||
| from services.normalization import get_stopwords, normalize_headline | |||||
| from services.sources import get_sources | |||||
| logging.basicConfig( | |||||
| level=logging.INFO, | |||||
| format='%(asctime)s [%(levelname)s] [%(name)s]: %(message)s' | |||||
| ) | |||||
| logger = logging.getLogger(__name__) | |||||
| SOURCE_FILE = './resources/sources.txt' | |||||
| STOPWORDS_FILE = './resources/stopwords.txt' | |||||
| SIMILARITY_THRESHOLD = 0.75 | |||||
| def main(): | |||||
| logger.info("Starting duplicate headline detection application") | |||||
| logger.debug("Source file configured at: '%s'", SOURCE_FILE) | |||||
| logger.debug("Stopwords file configured at: '%s'", STOPWORDS_FILE) | |||||
| try: | |||||
| logger.info("Loading news sources from '%s'", SOURCE_FILE) | |||||
| sources = get_sources(SOURCE_FILE) | |||||
| logger.info("Loaded %d sources successfully", len(sources) if sources else 0) | |||||
| except Exception as e: | |||||
| logger.critical("Failed to load sources from '%s': %s", SOURCE_FILE, e, exc_info=True) | |||||
| return | |||||
| try: | |||||
| logger.info("Loading stopwords from '%s'", STOPWORDS_FILE) | |||||
| stopwords = get_stopwords(STOPWORDS_FILE) | |||||
| logger.info("Loaded %d stopwords successfully", len(stopwords) if stopwords else 0) | |||||
| except Exception as e: | |||||
| logger.critical("Failed to load stopwords from '%s': %s", STOPWORDS_FILE, e, exc_info=True) | |||||
| return | |||||
| try: | |||||
| logger.info("Fetching and preparing headlines from %d sources", len(sources)) | |||||
| headlines = prepare_headlines(sources, stopwords) | |||||
| logger.info("Total prepared headlines available for comparison: %d", len(headlines)) | |||||
| except Exception as e: | |||||
| logger.critical("Failed during headline preparation: %s", e, exc_info=True) | |||||
| return | |||||
| total_comparisons = (len(headlines) * (len(headlines) - 1)) // 2 if len(headlines) > 1 else 0 | |||||
| logger.info("Beginning pairwise headline comparisons (total comparisons to execute: %d)", total_comparisons) | |||||
| duplicate_count = 0 | |||||
| comparison_idx = 0 | |||||
| for i in range(len(headlines)): | |||||
| for j in range(i + 1, len(headlines)): | |||||
| comparison_idx += 1 | |||||
| logger.debug("Comparison [%d/%d]: Headline %d vs Headline %d", comparison_idx, total_comparisons, i, j) | |||||
| try: | |||||
| similarity_score = headlines[i].compare_headlines(headlines[j]) | |||||
| logger.debug("Similarity score between [%d] and [%d] is %.4f (threshold: %f)", i, j, similarity_score, SIMILARITY_THRESHOLD) | |||||
| if similarity_score >= SIMILARITY_THRESHOLD: | |||||
| duplicate_count += 1 | |||||
| logger.warning("Duplicate/similar headline match found (score: %.4f < %f): '%s' vs '%s'", | |||||
| similarity_score, SIMILARITY_THRESHOLD, headlines[i].display_text, headlines[j].display_text) | |||||
| print(f"Duplicate headlines found: {headlines[i].display_text}") | |||||
| except Exception as e: | |||||
| logger.error("Error during comparison between headline %d (%r) and headline %d (%r): %s", | |||||
| i, headlines[i].display_text, j, headlines[j].display_text, e, exc_info=True) | |||||
| logger.info("Headline comparison completed. Evaluated %d pairs and found %d duplicate alerts.", | |||||
| comparison_idx, duplicate_count) | |||||
| if __name__ == '__main__': | |||||
| main() | |||||
| @@ -0,0 +1,20 @@ | |||||
| https://www.cnn.com/us | |||||
| https://www.cnn.com/politics | |||||
| https://www.foxnews.com/us | |||||
| https://www.foxnews.com/politics | |||||
| https://www.nbcnews.com/us-news | |||||
| https://www.nbcnews.com/politics | |||||
| https://www.reuters.com/world/us/ | |||||
| https://apnews.com/us-news | |||||
| https://apnews.com/politics | |||||
| https://www.npr.org/sections/national/ | |||||
| https://www.npr.org/sections/politics/ | |||||
| https://www.cbsnews.com/us/ | |||||
| https://www.cbsnews.com/politics/ | |||||
| https://www.forbes.com/business/ | |||||
| https://www.ms.now/ | |||||
| https://www.nytimes.com/section/us | |||||
| https://www.nytimes.com/section/politics | |||||
| https://www.usnews.com/news | |||||
| https://time.com/ | |||||
| https://www.axios.com/politics-policy | |||||
| @@ -0,0 +1,639 @@ | |||||
| a | |||||
| a's | |||||
| able | |||||
| about | |||||
| above | |||||
| abroad | |||||
| according | |||||
| accordingly | |||||
| across | |||||
| actually | |||||
| adj | |||||
| after | |||||
| afterwards | |||||
| again | |||||
| against | |||||
| ago | |||||
| ahead | |||||
| ain't | |||||
| all | |||||
| allow | |||||
| allows | |||||
| almost | |||||
| alone | |||||
| along | |||||
| alongside | |||||
| already | |||||
| also | |||||
| although | |||||
| always | |||||
| am | |||||
| amid | |||||
| amidst | |||||
| among | |||||
| amongst | |||||
| an | |||||
| and | |||||
| another | |||||
| any | |||||
| anybody | |||||
| anyhow | |||||
| anyone | |||||
| anything | |||||
| anyway | |||||
| anyways | |||||
| anywhere | |||||
| apart | |||||
| appear | |||||
| appreciate | |||||
| appropriate | |||||
| are | |||||
| aren't | |||||
| around | |||||
| as | |||||
| aside | |||||
| ask | |||||
| asking | |||||
| associated | |||||
| at | |||||
| available | |||||
| away | |||||
| awfully | |||||
| back | |||||
| backward | |||||
| backwards | |||||
| be | |||||
| became | |||||
| because | |||||
| become | |||||
| becomes | |||||
| becoming | |||||
| been | |||||
| before | |||||
| beforehand | |||||
| begin | |||||
| behind | |||||
| being | |||||
| believe | |||||
| below | |||||
| beside | |||||
| besides | |||||
| best | |||||
| better | |||||
| between | |||||
| beyond | |||||
| both | |||||
| brief | |||||
| but | |||||
| by | |||||
| c'mon | |||||
| c's | |||||
| came | |||||
| can | |||||
| can't | |||||
| cannot | |||||
| cant | |||||
| caption | |||||
| cause | |||||
| causes | |||||
| certain | |||||
| certainly | |||||
| changes | |||||
| clearly | |||||
| co | |||||
| co. | |||||
| com | |||||
| come | |||||
| comes | |||||
| concerning | |||||
| consequently | |||||
| consider | |||||
| considering | |||||
| contain | |||||
| containing | |||||
| contains | |||||
| corresponding | |||||
| could | |||||
| couldn't | |||||
| course | |||||
| currently | |||||
| dare | |||||
| daren't | |||||
| definitely | |||||
| described | |||||
| despite | |||||
| did | |||||
| didn't | |||||
| different | |||||
| directly | |||||
| do | |||||
| does | |||||
| doesn't | |||||
| doing | |||||
| don't | |||||
| done | |||||
| down | |||||
| downwards | |||||
| during | |||||
| each | |||||
| edu | |||||
| eg | |||||
| eight | |||||
| eighty | |||||
| either | |||||
| else | |||||
| elsewhere | |||||
| end | |||||
| ending | |||||
| enough | |||||
| entirely | |||||
| especially | |||||
| et | |||||
| etc | |||||
| even | |||||
| ever | |||||
| evermore | |||||
| every | |||||
| everybody | |||||
| everyone | |||||
| everything | |||||
| everywhere | |||||
| ex | |||||
| exactly | |||||
| example | |||||
| except | |||||
| fairly | |||||
| far | |||||
| farther | |||||
| few | |||||
| fewer | |||||
| fifth | |||||
| first | |||||
| five | |||||
| followed | |||||
| following | |||||
| follows | |||||
| for | |||||
| forever | |||||
| former | |||||
| formerly | |||||
| forth | |||||
| forward | |||||
| found | |||||
| four | |||||
| from | |||||
| further | |||||
| furthermore | |||||
| get | |||||
| gets | |||||
| getting | |||||
| given | |||||
| gives | |||||
| go | |||||
| goes | |||||
| going | |||||
| gone | |||||
| got | |||||
| gotten | |||||
| greetings | |||||
| had | |||||
| hadn't | |||||
| half | |||||
| happens | |||||
| hardly | |||||
| has | |||||
| hasn't | |||||
| have | |||||
| haven't | |||||
| having | |||||
| he | |||||
| he'd | |||||
| he'll | |||||
| he's | |||||
| hello | |||||
| help | |||||
| hence | |||||
| her | |||||
| here | |||||
| here's | |||||
| hereafter | |||||
| hereby | |||||
| herein | |||||
| hereupon | |||||
| hers | |||||
| herself | |||||
| hi | |||||
| him | |||||
| himself | |||||
| his | |||||
| hither | |||||
| hopefully | |||||
| how | |||||
| how's | |||||
| howbeit | |||||
| however | |||||
| hundred | |||||
| i | |||||
| i'd | |||||
| i'll | |||||
| i'm | |||||
| i've | |||||
| ie | |||||
| if | |||||
| ignored | |||||
| immediate | |||||
| in | |||||
| inasmuch | |||||
| inc | |||||
| inc. | |||||
| indeed | |||||
| indicate | |||||
| indicated | |||||
| indicates | |||||
| inner | |||||
| inside | |||||
| insofar | |||||
| instead | |||||
| into | |||||
| inward | |||||
| is | |||||
| isn't | |||||
| it | |||||
| it'd | |||||
| it'll | |||||
| it's | |||||
| its | |||||
| itself | |||||
| just | |||||
| k | |||||
| keep | |||||
| keeps | |||||
| kept | |||||
| know | |||||
| known | |||||
| knows | |||||
| last | |||||
| lately | |||||
| later | |||||
| latter | |||||
| latterly | |||||
| least | |||||
| less | |||||
| lest | |||||
| let | |||||
| let's | |||||
| like | |||||
| liked | |||||
| likely | |||||
| likewise | |||||
| little | |||||
| look | |||||
| looking | |||||
| looks | |||||
| low | |||||
| lower | |||||
| ltd | |||||
| made | |||||
| mainly | |||||
| make | |||||
| makes | |||||
| many | |||||
| may | |||||
| maybe | |||||
| mayn't | |||||
| me | |||||
| mean | |||||
| meantime | |||||
| meanwhile | |||||
| merely | |||||
| might | |||||
| mightn't | |||||
| mine | |||||
| minus | |||||
| miss | |||||
| more | |||||
| moreover | |||||
| most | |||||
| mostly | |||||
| mr | |||||
| mrs | |||||
| much | |||||
| must | |||||
| mustn't | |||||
| my | |||||
| myself | |||||
| name | |||||
| namely | |||||
| nd | |||||
| near | |||||
| nearly | |||||
| necessary | |||||
| need | |||||
| needn't | |||||
| needs | |||||
| neither | |||||
| never | |||||
| neverf | |||||
| neverless | |||||
| nevertheless | |||||
| new | |||||
| next | |||||
| nine | |||||
| ninety | |||||
| no | |||||
| no-one | |||||
| nobody | |||||
| non | |||||
| none | |||||
| nonetheless | |||||
| noone | |||||
| nor | |||||
| normally | |||||
| not | |||||
| nothing | |||||
| notwithstanding | |||||
| novel | |||||
| now | |||||
| nowhere | |||||
| obviously | |||||
| of | |||||
| off | |||||
| often | |||||
| oh | |||||
| ok | |||||
| okay | |||||
| old | |||||
| on | |||||
| once | |||||
| one | |||||
| one's | |||||
| ones | |||||
| only | |||||
| onto | |||||
| opposite | |||||
| or | |||||
| other | |||||
| others | |||||
| otherwise | |||||
| ought | |||||
| oughtn't | |||||
| our | |||||
| ours | |||||
| ourselves | |||||
| out | |||||
| outside | |||||
| over | |||||
| overall | |||||
| own | |||||
| particular | |||||
| particularly | |||||
| past | |||||
| per | |||||
| perhaps | |||||
| placed | |||||
| please | |||||
| plus | |||||
| possible | |||||
| presumably | |||||
| probably | |||||
| provided | |||||
| provides | |||||
| que | |||||
| quite | |||||
| qv | |||||
| rather | |||||
| rd | |||||
| re | |||||
| really | |||||
| reasonably | |||||
| recent | |||||
| recently | |||||
| regarding | |||||
| regardless | |||||
| regards | |||||
| relatively | |||||
| respectively | |||||
| right | |||||
| round | |||||
| said | |||||
| same | |||||
| saw | |||||
| say | |||||
| saying | |||||
| says | |||||
| second | |||||
| secondly | |||||
| see | |||||
| seeing | |||||
| seem | |||||
| seemed | |||||
| seeming | |||||
| seems | |||||
| seen | |||||
| self | |||||
| selves | |||||
| sensible | |||||
| sent | |||||
| serious | |||||
| seriously | |||||
| seven | |||||
| several | |||||
| shall | |||||
| shan't | |||||
| she | |||||
| she'd | |||||
| she'll | |||||
| she's | |||||
| should | |||||
| shouldn't | |||||
| since | |||||
| six | |||||
| so | |||||
| some | |||||
| somebody | |||||
| someday | |||||
| somehow | |||||
| someone | |||||
| something | |||||
| sometime | |||||
| sometimes | |||||
| somewhat | |||||
| somewhere | |||||
| soon | |||||
| sorry | |||||
| specified | |||||
| specify | |||||
| specifying | |||||
| still | |||||
| sub | |||||
| such | |||||
| sup | |||||
| sure | |||||
| t's | |||||
| take | |||||
| taken | |||||
| taking | |||||
| tell | |||||
| tends | |||||
| th | |||||
| than | |||||
| thank | |||||
| thanks | |||||
| thanx | |||||
| that | |||||
| that'll | |||||
| that's | |||||
| that've | |||||
| thats | |||||
| the | |||||
| their | |||||
| theirs | |||||
| them | |||||
| themselves | |||||
| then | |||||
| thence | |||||
| there | |||||
| there'd | |||||
| there'll | |||||
| there're | |||||
| there's | |||||
| there've | |||||
| thereafter | |||||
| thereby | |||||
| therefore | |||||
| therein | |||||
| theres | |||||
| thereupon | |||||
| these | |||||
| they | |||||
| they'd | |||||
| they'll | |||||
| they're | |||||
| they've | |||||
| thing | |||||
| things | |||||
| think | |||||
| third | |||||
| thirty | |||||
| this | |||||
| thorough | |||||
| thoroughly | |||||
| those | |||||
| though | |||||
| three | |||||
| through | |||||
| throughout | |||||
| thru | |||||
| thus | |||||
| till | |||||
| to | |||||
| together | |||||
| too | |||||
| took | |||||
| toward | |||||
| towards | |||||
| tried | |||||
| tries | |||||
| truly | |||||
| try | |||||
| trying | |||||
| twice | |||||
| two | |||||
| un | |||||
| under | |||||
| underneath | |||||
| undoing | |||||
| unfortunately | |||||
| unless | |||||
| unlike | |||||
| unlikely | |||||
| until | |||||
| unto | |||||
| up | |||||
| upon | |||||
| upwards | |||||
| use | |||||
| used | |||||
| useful | |||||
| uses | |||||
| using | |||||
| usually | |||||
| v | |||||
| value | |||||
| various | |||||
| versus | |||||
| very | |||||
| via | |||||
| viz | |||||
| vs | |||||
| want | |||||
| wants | |||||
| was | |||||
| wasn't | |||||
| way | |||||
| we | |||||
| we'd | |||||
| we'll | |||||
| we're | |||||
| we've | |||||
| welcome | |||||
| well | |||||
| went | |||||
| were | |||||
| weren't | |||||
| what | |||||
| what'll | |||||
| what's | |||||
| what've | |||||
| whatever | |||||
| when | |||||
| when's | |||||
| whence | |||||
| whenever | |||||
| where | |||||
| where's | |||||
| whereafter | |||||
| whereas | |||||
| whereby | |||||
| wherein | |||||
| whereupon | |||||
| wherever | |||||
| whether | |||||
| which | |||||
| whichever | |||||
| while | |||||
| whilst | |||||
| whither | |||||
| who | |||||
| who'd | |||||
| who'll | |||||
| who's | |||||
| whoever | |||||
| whole | |||||
| whom | |||||
| whomever | |||||
| whose | |||||
| why | |||||
| why's | |||||
| will | |||||
| willing | |||||
| wish | |||||
| with | |||||
| within | |||||
| without | |||||
| won't | |||||
| wonder | |||||
| would | |||||
| wouldn't | |||||
| yes | |||||
| yet | |||||
| you | |||||
| you'd | |||||
| you'll | |||||
| you're | |||||
| you've | |||||
| your | |||||
| yours | |||||
| yourself | |||||
| yourselves | |||||
| zero | |||||
| @@ -0,0 +1,94 @@ | |||||
| import logging | |||||
| from re import findall | |||||
| import requests | |||||
| from services.normalization import normalize_headline | |||||
| from structs.headline import Headline | |||||
| logger = logging.getLogger(__name__) | |||||
| DEFAULT_TIMEOUT = 10 | |||||
| MIN_HEADLINE_WORDS = 3 | |||||
| def is_headline(text, stopwords=None): | |||||
| if not text or not isinstance(text, str): | |||||
| return False | |||||
| cleaned_text = text.strip() | |||||
| if not cleaned_text: | |||||
| return False | |||||
| words = cleaned_text.split() | |||||
| if len(words) < MIN_HEADLINE_WORDS: | |||||
| logger.debug("Text rejected as headline (fewer than %d words): %r", MIN_HEADLINE_WORDS, cleaned_text) | |||||
| return False | |||||
| if not any(c.isalnum() for c in cleaned_text): | |||||
| logger.debug("Text rejected as headline (no alphanumeric characters): %r", cleaned_text) | |||||
| return False | |||||
| if stopwords is not None: | |||||
| normalized = normalize_headline(cleaned_text, stopwords) | |||||
| if not normalized: | |||||
| logger.debug("Text rejected as headline (no meaningful tokens after stopword removal): %r", cleaned_text) | |||||
| return False | |||||
| return True | |||||
| def prepare_headlines(sources, stopwords, timeout=DEFAULT_TIMEOUT): | |||||
| logger.info("Starting preparation of headlines for %d sources", len(sources) if sources else 0) | |||||
| headlines = [] | |||||
| if not sources: | |||||
| logger.warning("No sources provided to prepare_headlines.") | |||||
| return headlines | |||||
| for idx, source in enumerate(sources, start=1): | |||||
| if not source or not source.strip(): | |||||
| logger.warning("Skipping empty source at index %d", idx) | |||||
| continue | |||||
| source_url = source.strip() | |||||
| logger.info("Fetching source [%d/%d]: '%s'", idx, len(sources), source_url) | |||||
| try: | |||||
| response = requests.get(source_url, allow_redirects=True, timeout=timeout, headers={'User-Agent': 'Anya news bot'}) | |||||
| logger.debug("Received HTTP response %d for '%s' (content length: %d bytes)", | |||||
| response.status_code, source_url, len(response.content)) | |||||
| if response.status_code != 200: | |||||
| logger.warning("Source '%s' returned non-200 status code: %d", source_url, response.status_code) | |||||
| source_content = response.text | |||||
| except requests.exceptions.Timeout as e: | |||||
| logger.error("Request timed out for source '%s': %s", source_url, e, exc_info=True) | |||||
| continue | |||||
| except requests.exceptions.RequestException as e: | |||||
| logger.error("HTTP request failed for source '%s': %s", source_url, e, exc_info=True) | |||||
| continue | |||||
| except Exception as e: | |||||
| logger.error("Unexpected error fetching source '%s': %s", source_url, e, exc_info=True) | |||||
| continue | |||||
| logger.debug("Parsing HTML content from '%s' for headline candidates", source_url) | |||||
| try: | |||||
| link_texts = findall(r'<(?:a|span)\b[^>]*>\s*([^<]*?)\s*</(?:a|span)>', source_content) | |||||
| logger.info("Found %d candidate tags in source '%s'", len(link_texts), source_url) | |||||
| except Exception as e: | |||||
| logger.error("Regex extraction failed on content from '%s': %s", source_url, e, exc_info=True) | |||||
| continue | |||||
| source_headlines_count = 0 | |||||
| for tag_idx, link_text in enumerate(link_texts, start=1): | |||||
| cleaned_text = link_text.strip() | |||||
| if not cleaned_text: | |||||
| logger.debug("Skipping empty tag text at position %d from '%s'", tag_idx, source_url) | |||||
| continue | |||||
| if not is_headline(cleaned_text, stopwords): | |||||
| logger.debug("Skipping non-headline tag text at position %d from '%s': %r", tag_idx, source_url, cleaned_text) | |||||
| continue | |||||
| logger.debug("Processing tag [%d/%d] from '%s': %r", tag_idx, len(link_texts), source_url, cleaned_text) | |||||
| try: | |||||
| normalized_headline = normalize_headline(cleaned_text, stopwords) | |||||
| headline = Headline(cleaned_text, normalized_headline) | |||||
| headlines.append(headline) | |||||
| source_headlines_count += 1 | |||||
| except Exception as e: | |||||
| logger.error("Failed to normalize/create headline for text %r from '%s': %s", cleaned_text, source_url, e, exc_info=True) | |||||
| logger.info("Successfully extracted %d headlines from source '%s'", source_headlines_count, source_url) | |||||
| logger.info("Finished preparing headlines. Total headlines collected across all sources: %d", len(headlines)) | |||||
| return headlines | |||||
| @@ -0,0 +1,56 @@ | |||||
| import logging | |||||
| import string | |||||
| logger = logging.getLogger(__name__) | |||||
| def get_stopwords(path): | |||||
| logger.info("Attempting to load stopwords from file: '%s'", path) | |||||
| try: | |||||
| with open(path, 'r', encoding='utf-8-sig') as stopwords_file: | |||||
| logger.debug("Opened stopwords file '%s'", path) | |||||
| lines = stopwords_file.read().splitlines() | |||||
| stopwords = set(lines) | |||||
| logger.info("Successfully loaded %d unique stopwords from '%s' (%d total lines)", len(stopwords), path, len(lines)) | |||||
| return stopwords | |||||
| except FileNotFoundError: | |||||
| logger.error("Stopwords file not found at path: '%s'", path, exc_info=True) | |||||
| raise | |||||
| except PermissionError: | |||||
| logger.error("Permission denied accessing stopwords file: '%s'", path, exc_info=True) | |||||
| raise | |||||
| except Exception as e: | |||||
| logger.error("Failed to load stopwords from '%s': %s", path, e, exc_info=True) | |||||
| raise | |||||
| def remove_stopwords(text, stopwords): | |||||
| logger.debug("Removing stopwords from text (%d chars): %r (available stopwords: %d)", len(text), text, len(stopwords)) | |||||
| # split() without arguments handles all whitespace (spaces, tabs, newlines) | |||||
| words = text.split() | |||||
| sentence_words = [] | |||||
| for word in words: | |||||
| # Strip surrounding punctuation and lowercase for comparison | |||||
| cleaned_word = word.strip(string.punctuation).lower() | |||||
| if cleaned_word and cleaned_word not in stopwords: | |||||
| sentence_words.append(word) | |||||
| elif cleaned_word in stopwords: | |||||
| logger.debug("Word filtered out as stopword: %r (original: %r)", cleaned_word, word) | |||||
| else: | |||||
| logger.debug("Word dropped (empty after stripping punctuation): %r", word) | |||||
| logger.debug("Filtered words: input=%d words, output=%d words -> %s", len(words), len(sentence_words), sentence_words) | |||||
| return sentence_words | |||||
| def normalize_headline(text, stopwords): | |||||
| logger.debug("Starting normalization of headline: %r", text) | |||||
| headline = text.strip().lower() | |||||
| punctuation = string.punctuation | |||||
| for char in punctuation: | |||||
| headline = headline.replace(char, '') | |||||
| logger.debug("Headline after lowercasing and punctuation removal: %r", headline) | |||||
| normalized = remove_stopwords(headline, stopwords) | |||||
| logger.debug("Completed normalization for %r -> %s", text, normalized) | |||||
| return normalized | |||||
| @@ -0,0 +1,57 @@ | |||||
| from collections import Counter | |||||
| import logging | |||||
| from math import sqrt | |||||
| from collections.abc import Sequence | |||||
| logger = logging.getLogger(__name__) | |||||
| def cosine_similarity(a: Sequence[float], b: Sequence[float]) -> float: | |||||
| logger.debug("Computing cosine similarity between numerical vectors of length %d and %d", len(a), len(b)) | |||||
| if len(a) != len(b): | |||||
| logger.error("Vector dimension mismatch: vector 'a' length (%d) != vector 'b' length (%d)", len(a), len(b)) | |||||
| raise ValueError("Vectors must have the same dimension") | |||||
| dot = 0.0 | |||||
| norm_a_sq = 0.0 | |||||
| norm_b_sq = 0.0 | |||||
| for x, y in zip(a, b): | |||||
| dot += x * y | |||||
| norm_a_sq += x * x | |||||
| norm_b_sq += y * y | |||||
| denominator = sqrt(norm_a_sq * norm_b_sq) | |||||
| if denominator == 0.0: | |||||
| logger.debug("Zero denominator encountered in cosine_similarity (norm_a_sq=%f, norm_b_sq=%f). Returning 0.0", norm_a_sq, norm_b_sq) | |||||
| return 0.0 | |||||
| similarity = dot / denominator | |||||
| logger.debug("Calculated vector cosine similarity: dot=%f, denominator=%f, similarity=%f", dot, denominator, similarity) | |||||
| return similarity | |||||
| def cosine_lists(a: list[str], b: list[str], *, casefold: bool = True) -> float: | |||||
| logger.debug("Computing token cosine similarity for list_a=%s and list_b=%s (casefold=%s)", a, b, casefold) | |||||
| def tokens(xs: list[str]) -> Counter[str]: | |||||
| return Counter(x.casefold() if casefold else x for x in xs) | |||||
| ca, cb = tokens(a), tokens(b) | |||||
| if not ca or not cb: | |||||
| logger.debug("Empty token set detected (count_a=%d, count_b=%d). Cosine similarity is 0.0", len(ca), len(cb)) | |||||
| return 0.0 | |||||
| common_tokens = ca.keys() & cb.keys() | |||||
| dot = sum(ca[t] * cb[t] for t in common_tokens) | |||||
| norm_a = sqrt(sum(v * v for v in ca.values())) | |||||
| norm_b = sqrt(sum(v * v for v in cb.values())) | |||||
| if norm_a == 0.0 or norm_b == 0.0: | |||||
| logger.debug("Zero norm detected (norm_a=%f, norm_b=%f). Cosine similarity is 0.0", norm_a, norm_b) | |||||
| return 0.0 | |||||
| similarity = dot / (norm_a * norm_b) | |||||
| logger.debug("Token similarity calculation: common_tokens=%s, dot=%f, norm_a=%f, norm_b=%f -> similarity=%.4f", | |||||
| list(common_tokens), dot, norm_a, norm_b, similarity) | |||||
| return similarity | |||||
| @@ -0,0 +1,31 @@ | |||||
| import logging | |||||
| logger = logging.getLogger(__name__) | |||||
| def get_sources(path, delimiter='\n'): | |||||
| logger.info("Attempting to load sources from file: '%s' with delimiter: %r", path, delimiter) | |||||
| sources = None | |||||
| try: | |||||
| with open(path, 'r', encoding='utf-8') as source_file: | |||||
| logger.debug("Successfully opened file '%s' for reading", path) | |||||
| content = source_file.read() | |||||
| logger.debug("Read %d bytes/characters from '%s'", len(content), path) | |||||
| sources = content.split(delimiter) | |||||
| logger.info("Successfully read and split %d source entries from '%s'", len(sources), path) | |||||
| for idx, src in enumerate(sources): | |||||
| if not src.strip(): | |||||
| logger.warning("Source at index %d is empty or whitespace-only: %r", idx, src) | |||||
| else: | |||||
| logger.debug("Source [%d]: %s", idx, src) | |||||
| except FileNotFoundError: | |||||
| logger.error("Source file not found at path: '%s'", path, exc_info=True) | |||||
| raise | |||||
| except PermissionError: | |||||
| logger.error("Permission denied when accessing source file: '%s'", path, exc_info=True) | |||||
| raise | |||||
| except Exception as e: | |||||
| logger.error("Failed to read sources from '%s': %s", path, e, exc_info=True) | |||||
| raise | |||||
| return sources | |||||
| @@ -0,0 +1,32 @@ | |||||
| import logging | |||||
| from services.similarity import cosine_lists | |||||
| logger = logging.getLogger(__name__) | |||||
| class Headline: | |||||
| display_text: str | |||||
| normalized_text: list[str] | |||||
| def __init__(self, display_text: str, normalized_text: list[str]): | |||||
| self.display_text = display_text | |||||
| self.normalized_text = normalized_text | |||||
| logger.debug("Initialized Headline instance (display_text=%r, token_count=%d): %s", | |||||
| self.display_text, len(self.normalized_text), self.normalized_text) | |||||
| def compare_headlines(self, other_headline): | |||||
| if not isinstance(other_headline, Headline): | |||||
| logger.warning("Comparing Headline %r with incompatible object of type %s: %r", | |||||
| self.display_text, type(other_headline).__name__, other_headline) | |||||
| logger.debug("Comparing Headline %r against %r", | |||||
| self.display_text, getattr(other_headline, 'display_text', repr(other_headline))) | |||||
| try: | |||||
| score = cosine_lists(self.normalized_text, other_headline.normalized_text) | |||||
| logger.debug("Headline comparison score: %.4f between %r and %r", | |||||
| score, self.display_text, getattr(other_headline, 'display_text', repr(other_headline))) | |||||
| return score | |||||
| except Exception as e: | |||||
| logger.error("Error comparing headlines (%r vs %r): %s", | |||||
| self.display_text, getattr(other_headline, 'display_text', repr(other_headline)), e, exc_info=True) | |||||
| raise | |||||
| @@ -0,0 +1,154 @@ | |||||
| import unittest | |||||
| from unittest.mock import patch, MagicMock | |||||
| import requests | |||||
| from services.headlines import prepare_headlines, is_headline, DEFAULT_TIMEOUT | |||||
| class TestHeadlinesTimeout(unittest.TestCase): | |||||
| def setUp(self): | |||||
| self.stopwords = {"the", "a", "an", "in", "on"} | |||||
| @patch("services.headlines.requests.get") | |||||
| def test_default_timeout_used(self, mock_get): | |||||
| mock_response = MagicMock() | |||||
| mock_response.status_code = 200 | |||||
| mock_response.content = b"<html><body><a href='#'>Default Timeout Headline</a></body></html>" | |||||
| mock_response.text = "<html><body><a href='#'>Default Timeout Headline</a></body></html>" | |||||
| mock_get.return_value = mock_response | |||||
| sources = ["https://example.com/news"] | |||||
| headlines = prepare_headlines(sources, self.stopwords) | |||||
| mock_get.assert_called_once_with("https://example.com/news", allow_redirects=True, timeout=DEFAULT_TIMEOUT, headers={'User-Agent': 'Anya news bot'}) | |||||
| self.assertEqual(len(headlines), 1) | |||||
| self.assertEqual(headlines[0].display_text, "Default Timeout Headline") | |||||
| @patch("services.headlines.requests.get") | |||||
| def test_custom_timeout_used(self, mock_get): | |||||
| mock_response = MagicMock() | |||||
| mock_response.status_code = 200 | |||||
| mock_response.content = b"<html><body><a href='#'>Custom Timeout Headline</a></body></html>" | |||||
| mock_response.text = "<html><body><a href='#'>Custom Timeout Headline</a></body></html>" | |||||
| mock_get.return_value = mock_response | |||||
| sources = ["https://example.com/news"] | |||||
| headlines = prepare_headlines(sources, self.stopwords, timeout=10) | |||||
| mock_get.assert_called_once_with("https://example.com/news", allow_redirects=True, timeout=10, headers={'User-Agent': 'Anya news bot'}) | |||||
| self.assertEqual(len(headlines), 1) | |||||
| @patch("services.headlines.requests.get") | |||||
| def test_timeout_skips_slow_request_and_processes_others(self, mock_get): | |||||
| slow_url = "https://slow-source.example.com" | |||||
| fast_url = "https://fast-source.example.com" | |||||
| def side_effect(url, **kwargs): | |||||
| if url == slow_url: | |||||
| raise requests.exceptions.Timeout("Connection timed out after %s seconds" % kwargs.get("timeout")) | |||||
| fast_response = MagicMock() | |||||
| fast_response.status_code = 200 | |||||
| fast_response.content = b"<html><body><span>Breaking News Story</span></body></html>" | |||||
| fast_response.text = "<html><body><span>Breaking News Story</span></body></html>" | |||||
| return fast_response | |||||
| mock_get.side_effect = side_effect | |||||
| sources = [slow_url, fast_url] | |||||
| headlines = prepare_headlines(sources, self.stopwords, timeout=3) | |||||
| self.assertEqual(mock_get.call_count, 2) | |||||
| self.assertEqual(len(headlines), 1) | |||||
| self.assertEqual(headlines[0].display_text, "Breaking News Story") | |||||
| @patch("services.headlines.requests.get") | |||||
| def test_connect_timeout_and_read_timeout_skipped(self, mock_get): | |||||
| connect_timeout_url = "https://connect-timeout.com" | |||||
| read_timeout_url = "https://read-timeout.com" | |||||
| mock_get.side_effect = [ | |||||
| requests.exceptions.ConnectTimeout("Connect timeout"), | |||||
| requests.exceptions.ReadTimeout("Read timeout") | |||||
| ] | |||||
| sources = [connect_timeout_url, read_timeout_url] | |||||
| headlines = prepare_headlines(sources, self.stopwords) | |||||
| self.assertEqual(mock_get.call_count, 2) | |||||
| self.assertEqual(len(headlines), 0) | |||||
| class TestHeadlineSafeguards(unittest.TestCase): | |||||
| def setUp(self): | |||||
| self.stopwords = {"the", "a", "an", "in", "on", "and", "of", "to"} | |||||
| def test_non_headline_link_texts_rejected(self): | |||||
| non_headlines = [ | |||||
| "Account Settings", | |||||
| "Follow", | |||||
| "Television", | |||||
| "Sign In", | |||||
| "Home", | |||||
| "Politics", | |||||
| "About Us", | |||||
| "Contact Us", | |||||
| "Menu", | |||||
| "Search", | |||||
| "", | |||||
| " ", | |||||
| "123", | |||||
| "...", | |||||
| "in on a", # only stopwords | |||||
| ] | |||||
| for item in non_headlines: | |||||
| with self.subTest(item=item): | |||||
| self.assertFalse(is_headline(item, self.stopwords), f"{item!r} should not be recognized as a headline") | |||||
| def test_valid_headlines_accepted(self): | |||||
| valid_headlines = [ | |||||
| "Breaking News Story", | |||||
| "Custom Timeout Headline", | |||||
| "Senate passes major infrastructure bill", | |||||
| "Scientists discover new ocean species", | |||||
| "Federal Reserve holds interest rates steady", | |||||
| ] | |||||
| for item in valid_headlines: | |||||
| with self.subTest(item=item): | |||||
| self.assertTrue(is_headline(item, self.stopwords), f"{item!r} should be recognized as a headline") | |||||
| @patch("services.headlines.requests.get") | |||||
| def test_prepare_headlines_filters_out_navigation_and_non_headlines(self, mock_get): | |||||
| html_content = """ | |||||
| <html> | |||||
| <body> | |||||
| <a href="/settings">Account Settings</a> | |||||
| <a href="/social">Follow</a> | |||||
| <a href="/tv">Television</a> | |||||
| <a href="/login">Sign In</a> | |||||
| <a href="/article1">Senate passes major infrastructure bill</a> | |||||
| <span>Home</span> | |||||
| <span>Federal Reserve holds interest rates steady</span> | |||||
| </body> | |||||
| </html> | |||||
| """ | |||||
| mock_response = MagicMock() | |||||
| mock_response.status_code = 200 | |||||
| mock_response.content = html_content.encode("utf-8") | |||||
| mock_response.text = html_content | |||||
| mock_get.return_value = mock_response | |||||
| sources = ["https://example.com/news"] | |||||
| headlines = prepare_headlines(sources, self.stopwords) | |||||
| self.assertEqual(len(headlines), 2) | |||||
| extracted_texts = [h.display_text for h in headlines] | |||||
| self.assertIn("Senate passes major infrastructure bill", extracted_texts) | |||||
| self.assertIn("Federal Reserve holds interest rates steady", extracted_texts) | |||||
| self.assertNotIn("Account Settings", extracted_texts) | |||||
| self.assertNotIn("Follow", extracted_texts) | |||||
| self.assertNotIn("Television", extracted_texts) | |||||
| self.assertNotIn("Sign In", extracted_texts) | |||||
| self.assertNotIn("Home", extracted_texts) | |||||
| if __name__ == "__main__": | |||||
| unittest.main() | |||||