This repository was archived by the owner on May 26, 2026. It is now read-only.
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathmain_local.py
More file actions
70 lines (61 loc) · 2.13 KB
/
Copy pathmain_local.py
File metadata and controls
70 lines (61 loc) · 2.13 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
import json
import timeit
from models import Source
from scrapy.crawler import CrawlerProcess, Settings
from scrapy_crawler import settings as local_crawler_settings
from scrapy_crawler.pipelines import NewsCrawlerPipeline
from scrapy_crawler.spiders.nld import NguoiLaoDongSpider
from scrapy_crawler.spiders.tuoitre import TuoiTreSpider
from scrapy_crawler.spiders.vnexpress import VnExpressSpider
def main_local(*args, **kwargs):
start_time = timeit.default_timer()
print(">> Start crawling...", flush=True)
NewsCrawlerPipeline.save_spider_articles = True
spiders = [
TuoiTreSpider,
VnExpressSpider,
NguoiLaoDongSpider,
]
crawler_settings = Settings()
crawler_settings.setmodule(local_crawler_settings)
crawler = CrawlerProcess(settings=crawler_settings)
sources = {
"nld": Source(
editor_id="nld",
urls={
"business": "https://nld.com.vn/kinh-te.rss",
},
),
"vnexpress": Source(
editor_id="vnexpress",
urls={
"business": "https://vnexpress.net/rss/kinh-doanh.rss",
},
),
"tuoitre": Source(
editor_id="tuoitre",
urls={
"business": "https://tuoitre.vn/rss/kinh-doanh.rss",
},
),
}
for spider in spiders:
crawler.crawl(spider, sources[spider.name])
crawler.join()
crawler.start()
number_of_articles = 0
for topic in NewsCrawlerPipeline.articles_by_topics:
number_of_articles += len(NewsCrawlerPipeline.articles_by_topics[topic].keys())
print(f"> {topic}: {len(NewsCrawlerPipeline.articles_by_topics[topic].keys())}")
elapsed_time = round(timeit.default_timer() - start_time, 4)
print(f">> Number of articles: {number_of_articles}")
print(f">> Time elapsed: {elapsed_time}s")
with open("crawled_data/articles_by_topics.json", "w", encoding="utf-8") as file:
json.dump(
NewsCrawlerPipeline.articles_by_topics,
file,
ensure_ascii=False,
indent=2,
default=str,
)
main_local()