{"id":21852184,"url":"https://github.com/brucedone/scrapy_demo","last_synced_at":"2025-05-08T21:23:42.192Z","repository":{"id":2368444,"uuid":"46345679","full_name":"BruceDone/scrapy_demo","owner":"BruceDone","description":"all kinds of scrapy demo ","archived":false,"fork":false,"pushed_at":"2023-01-31T11:24:21.000Z","size":69,"stargazers_count":164,"open_issues_count":2,"forks_count":57,"subscribers_count":8,"default_branch":"master","last_synced_at":"2025-02-02T03:51:56.429Z","etag":null,"topics":["cnbeta","demo","douban-image","example","imagespipeline","kafak","kafka","mongodb","oss","pipeline","scrapy","scrapy-demo","spider","sqlalchemy"],"latest_commit_sha":null,"homepage":"http://brucedone.com","language":"Python","has_issues":true,"has_wiki":null,"has_pages":null,"mirror_url":null,"source_name":null,"license":null,"status":null,"scm":"git","pull_requests_enabled":true,"icon_url":"https://github.com/BruceDone.png","metadata":{"files":{"readme":"README.md","changelog":null,"contributing":null,"funding":null,"license":null,"code_of_conduct":null,"threat_model":null,"audit":null,"citation":null,"codeowners":null,"security":null,"support":null}},"created_at":"2015-11-17T12:28:46.000Z","updated_at":"2025-01-20T13:33:02.000Z","dependencies_parsed_at":"2023-02-16T18:45:44.179Z","dependency_job_id":null,"html_url":"https://github.com/BruceDone/scrapy_demo","commit_stats":null,"previous_names":[],"tags_count":0,"template":false,"template_full_name":null,"repository_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/BruceDone%2Fscrapy_demo","tags_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/BruceDone%2Fscrapy_demo/tags","releases_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/BruceDone%2Fscrapy_demo/releases","manifests_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories/BruceDone%2Fscrapy_demo/manifests","owner_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/owners/BruceDone","download_url":"https://codeload.github.com/BruceDone/scrapy_demo/tar.gz/refs/heads/master","host":{"name":"GitHub","url":"https://github.com","kind":"github","repositories_count":238044094,"owners_count":19407128,"icon_url":"https://github.com/github.png","version":null,"created_at":"2022-05-30T11:31:42.601Z","updated_at":"2022-07-04T15:15:14.044Z","host_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub","repositories_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repositories","repository_names_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/repository_names","owners_url":"https://repos.ecosyste.ms/api/v1/hosts/GitHub/owners"}},"keywords":["cnbeta","demo","douban-image","example","imagespipeline","kafak","kafka","mongodb","oss","pipeline","scrapy","scrapy-demo","spider","sqlalchemy"],"created_at":"2024-11-28T01:14:15.666Z","updated_at":"2025-02-10T02:09:20.151Z","avatar_url":"https://github.com/BruceDone.png","language":"Python","funding_links":[],"categories":[],"sub_categories":[],"readme":"# Scrapy_demo\nthis project scrapes a list of websites I used to crawl most often\nif this project helped you, please give it a star, thanks :)\n\n# Spider list\n* douban\n* douban_oss\n* googleplay\n* cnbeta\n* ka\n* cnblogs\n\n# Project Feature\n* `google play` uses the crawl spider and pymongo\n* `douban` use the images pipeline to download image (use the headers in case of being banned), after finish it will output the txt file of item information\n* `cnbeta` uses sqlalchmey to save items to mysql database (or other database if sqlalchemy supports)\n* `ka` uses the kafka , this is a demo spider how to use the scrapy and kafka together , this spider will not close , if you push a message to the kafka ,the spider will start to crawl the url you just give\n* `cnblogs` use the signal handler.\n* `douban_oss` use the aliyun oss sdk upload the images pipeline download image to oss store. \n\n# How to use\nfor each project there is a run_spider.py script, just run it and enjoy :)\n\n```\npython run_spider.py\n```\n","project_url":"https://awesome.ecosyste.ms/api/v1/projects/github.com%2Fbrucedone%2Fscrapy_demo","html_url":"https://awesome.ecosyste.ms/projects/github.com%2Fbrucedone%2Fscrapy_demo","lists_url":"https://awesome.ecosyste.ms/api/v1/projects/github.com%2Fbrucedone%2Fscrapy_demo/lists"}