{"id":15208828,"url":"https://github.com/soheil-mp/sales-analytics-pipeline","last_synced_at":"2026-01-27T11:33:14.719Z","repository":{"id":248918935,"uuid":"830181582","full_name":"soheil-mp/Sales-Analytics-Pipeline","owner":"soheil-mp","description":"Data analytics pipeline built with Apache Spark and Hadoop for processing and analyzing large-scale sales data.","archived":false,"fork":false,"pushed_at":"2024-07-17T19:31:14.000Z","size":6,"stargazers_count":0,"open_issues_count":0,"forks_count":0,"subscribers_count":1,"default_branch":"master","last_synced_at":"2025-03-06T03:33:51.467Z","etag":null,"topics":["apache-spark","hadoop","hdfs","sql"],"latest_commit_sha":null,"homepage":"","language":"Python","has_issues":true,"has_wiki":null,"has_pages":null,"mirror_url":null,"source_name":null,"license":null,"status":null,"scm":"git","pull_requests_enabled":true,"icon_url":"https://github.com/soheil-mp.png","metadata":{"files":{"readme":"README.md","changelog":null,"contributing":null,"funding":null,"license":null,"code_of_conduct":null,"threat_model":null,"audit":null,"citation":null,"codeowners":null,"security":null,"support":null,"governance":null,"roadmap":null,"authors":null,"dei":null,"publiccode":null,"codemeta":null}},"created_at":"2024-07-17T18:57:54.000Z","updated_at":"2024-07-17T19:31:17.000Z","dependencies_parsed_at":"2024-07-17T23:17:22.585Z","dependency_job_id":null,"html_url":"https://github.com/soheil-mp/Sales-Analytics-Pipeline","commit_stats":null,"previous_names":["soheil-mp/sales-analytics-pipeline"],"tags_count":0,"template":false,"template_full_name":null,"repository_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/soheil-mp%2FSales-Analytics-Pipeline","tags_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/soheil-mp%2FSales-Analytics-Pipeline/tags","releases_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/soheil-mp%2FSales-Analytics-Pipeline/releases","manifests_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/soheil-mp%2FSales-Analytics-Pipeline/manifests","owner_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/owners/soheil-mp","download_url":"https://codeload.github.com/soheil-mp/Sales-Analytics-Pipeline/tar.gz/refs/heads/master","host":{"name":"GitHub","url":"https://github.com","kind":"github","repositories_count":243023500,"owners_count":20223432,"icon_url":"https://github.com/github.png","version":null,"created_at":"2022-05-30T11:31:42.601Z","updated_at":"2022-07-04T15:15:14.044Z","host_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub","repositories_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories","repository_names_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repository_names","owners_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/owners"}},"keywords":["apache-spark","hadoop","hdfs","sql"],"created_at":"2024-09-28T07:02:13.166Z","updated_at":"2026-01-27T11:33:14.671Z","avatar_url":"https://github.com/soheil-mp.png","language":"Python","funding_links":[],"categories":[],"sub_categories":[],"readme":"# Sales-Analytics-Pipeline\n\nA comprehensive data analytics pipeline built with Apache Spark and Hadoop for processing and analyzing large-scale sales data.\n\nThis project demonstrates how to read and write data from and to HDFS, clean and preprocess data using PySpark, conduct advanced analytics with Spark SQL and window functions, integrate with Hive for data warehousing, and maintain a modular code structure for complex ETL processes.\n\nKey features include scalable sales data processing, monthly sales trend analysis, insights into customer purchasing behavior, evaluation of product category performance, configurable data input/output paths, and robust error handling and logging.\n\nThe tech stack utilized in this project comprises Apache Spark, the Hadoop Distributed File System (HDFS), Apache Hive, and Python.\n\nRun the following command to start the codes:\n```bash\n$ spark-submit --master yarn \\\n               --deploy-mode client \\\n               --driver-memory 2g \\\n               --executor-memory 2g \\\n               --executor-cores 2 \\\n               main.py\n```\n","project_url":"https://awesome.ecosyste.ms/api/v1/projects/github.com%2Fsoheil-mp%2Fsales-analytics-pipeline","html_url":"https://awesome.ecosyste.ms/projects/github.com%2Fsoheil-mp%2Fsales-analytics-pipeline","lists_url":"https://awesome.ecosyste.ms/api/v1/projects/github.com%2Fsoheil-mp%2Fsales-analytics-pipeline/lists"}