{"id":30462822,"url":"https://github.com/prab9166/-web-content-extraction-and-sentiment-analysis-","last_synced_at":"2026-05-06T17:34:16.654Z","repository":{"id":309111390,"uuid":"1035202425","full_name":"prab9166/-web-content-extraction-and-sentiment-analysis-","owner":"prab9166","description":" web content extraction and sentiment analysis on URLs provided in an Excel file. Its divided into classes to manage different stages such as data loading, web scraping, text processing, sentiment scoring, and output generation","archived":false,"fork":false,"pushed_at":"2026-03-11T23:21:32.000Z","size":7,"stargazers_count":0,"open_issues_count":0,"forks_count":0,"subscribers_count":0,"default_branch":"main","last_synced_at":"2026-03-12T04:32:02.437Z","etag":null,"topics":["beautifulsoup","lxml","nltk-python","numpy","pandas","python","re","requests","xlsxwriter"],"latest_commit_sha":null,"homepage":"","language":"Python","has_issues":true,"has_wiki":null,"has_pages":null,"mirror_url":null,"source_name":null,"license":null,"status":null,"scm":"git","pull_requests_enabled":true,"icon_url":"https://github.com/prab9166.png","metadata":{"files":{"readme":"README.md","changelog":null,"contributing":null,"funding":null,"license":null,"code_of_conduct":null,"threat_model":null,"audit":null,"citation":null,"codeowners":null,"security":null,"support":null,"governance":null,"roadmap":null,"authors":null,"dei":null,"publiccode":null,"codemeta":null,"zenodo":null}},"created_at":"2025-08-09T21:52:17.000Z","updated_at":"2026-03-11T23:21:36.000Z","dependencies_parsed_at":"2025-08-09T23:44:27.415Z","dependency_job_id":null,"html_url":"https://github.com/prab9166/-web-content-extraction-and-sentiment-analysis-","commit_stats":null,"previous_names":["prab9166/-web-content-extraction-and-sentiment-analysis-"],"tags_count":0,"template":false,"template_full_name":null,"purl":"pkg:github/prab9166/-web-content-extraction-and-sentiment-analysis-","repository_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/prab9166%2F-web-content-extraction-and-sentiment-analysis-","tags_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/prab9166%2F-web-content-extraction-and-sentiment-analysis-/tags","releases_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/prab9166%2F-web-content-extraction-and-sentiment-analysis-/releases","manifests_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/prab9166%2F-web-content-extraction-and-sentiment-analysis-/manifests","owner_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/owners/prab9166","download_url":"https://codeload.github.com/prab9166/-web-content-extraction-and-sentiment-analysis-/tar.gz/refs/heads/main","sbom_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/prab9166%2F-web-content-extraction-and-sentiment-analysis-/sbom","scorecard":null,"host":{"name":"GitHub","url":"https://github.com","kind":"github","repositories_count":286080680,"owners_count":32704499,"icon_url":"https://github.com/github.png","version":null,"created_at":"2022-05-30T11:31:42.601Z","updated_at":"2026-05-06T08:33:17.875Z","status":"ssl_error","status_checked_at":"2026-05-06T08:33:17.221Z","response_time":117,"last_error":"SSL_connect returned=1 errno=0 peeraddr=140.82.121.6:443 state=error: unexpected eof while reading","robots_txt_status":"success","robots_txt_updated_at":"2025-07-24T06:49:26.215Z","robots_txt_url":"https://github.com/robots.txt","online":false,"can_crawl_api":true,"host_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub","repositories_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories","repository_names_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repository_names","owners_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/owners"}},"keywords":["beautifulsoup","lxml","nltk-python","numpy","pandas","python","re","requests","xlsxwriter"],"created_at":"2025-08-23T23:02:37.302Z","updated_at":"2026-05-06T17:34:16.650Z","avatar_url":"https://github.com/prab9166.png","language":"Python","funding_links":[],"categories":[],"sub_categories":[],"readme":"# Web Content Extraction and Sentiment Analysis\n\n## Overview\n\nThis project extracts textual content from a list of website URLs and performs sentiment analysis on the extracted text.\n\nThe goal was to build a simple workflow that automatically collects web content and analyzes whether the text expresses positive, negative, or neutral sentiment.\n\nThis type of workflow can be useful for analyzing news articles, blogs, or online content at scale.\n\n---\n\n## What the Project Does\n\nThe script performs the following steps:\n\n1. Reads a dataset containing website URLs\n2. Extracts the webpage text content\n3. Cleans and processes the extracted text\n4. Applies sentiment analysis\n5. Outputs structured results with sentiment scores\n\n---\n\n## Technologies Used\n\n- Python\n- Pandas\n- BeautifulSoup\n- Natural Language Processing (NLP)\n\n---\n\n## Example Use Case\n\nThis workflow can be used for:\n\n- media sentiment tracking\n- brand monitoring\n- automated content analysis\n- research data preparation\n\n---\n\n## How to Run the Project\n\nClone the repository\n","project_url":"https://awesome.ecosyste.ms/api/v1/projects/github.com%2Fprab9166%2F-web-content-extraction-and-sentiment-analysis-","html_url":"https://awesome.ecosyste.ms/projects/github.com%2Fprab9166%2F-web-content-extraction-and-sentiment-analysis-","lists_url":"https://awesome.ecosyste.ms/api/v1/projects/github.com%2Fprab9166%2F-web-content-extraction-and-sentiment-analysis-/lists"}