diff --git a/hako-crawler/.gitignore b/hako-crawler/.gitignore new file mode 100644 index 0000000..723ef36 --- /dev/null +++ b/hako-crawler/.gitignore @@ -0,0 +1 @@ +.idea \ No newline at end of file diff --git a/hako-crawler/README.md b/hako-crawler/README.md index 45da7e3..afa2f88 100644 --- a/hako-crawler/README.md +++ b/hako-crawler/README.md @@ -1,2 +1,18 @@ # atnvc-crawler -Crawl data of ATNVC + +Crawl data of "Anh trai nhân vật chính", a novel from https://ln.hako.vn/sang-tac/8476-kiep-nay-la-anh-trai-cua-nhan-vat-chinh + +# How to run +1. Install requirements + ``` + pip install -r requirements.txt + ``` +2. Run main.py to crawl data + ``` + python main.py + ``` + __Note__: You may get `HTTP Error 429: Too Many Requests`. Then you can try again later, and skip downloaded chapters, example skip 50 first chapters like: + ``` + for chapter in chapters[50:]: + ``` +3. The data will be saved in `data` folder \ No newline at end of file diff --git a/hako-crawler/data/.gitignore b/hako-crawler/data/.gitignore new file mode 100644 index 0000000..c96a04f --- /dev/null +++ b/hako-crawler/data/.gitignore @@ -0,0 +1,2 @@ +* +!.gitignore \ No newline at end of file diff --git a/hako-crawler/main.py b/hako-crawler/main.py new file mode 100644 index 0000000..39859e9 --- /dev/null +++ b/hako-crawler/main.py @@ -0,0 +1,54 @@ +import re +from urllib.request import Request, urlopen +from bs4 import BeautifulSoup + +DEBUG = False + + +def get_from_url(url): + request_site = Request(url, headers={"User-Agent": "Mozilla/5.0"}) + webpage = urlopen(request_site).read() + return webpage.decode("utf-8") + + +def write_text_to_file(text, filename): + filename = re.sub(r"[^\w_. -]", "_", filename) + f = open("data/{}".format(filename), "w", encoding="utf-8") + f.write(text) + f.close() + + +def read_text_from_file(filename): + f = open("data/{}".format(filename), "r", encoding="utf-8") + text = f.read() + f.close() + return text + + +if DEBUG: + html = read_text_from_file("_.txt") +else: + html = get_from_url("https://ln.hako.vn/sang-tac/8476-kiep-nay-la-anh-trai-cua-nhan-vat-chinh") + # write_text_to_file(html, "_.txt") + +soup = BeautifulSoup(html, "html.parser") +chapters = soup.find_all("div", {"class": "chapter-name"}) +""" +you may get `HTTP Error 429: Too Many Requests` +you can try again later, and skip downloaded chapters +example skip 50 first chapters like: `for chapter in chapters[50:]` +""" +for chapter in chapters: + children = chapter.find_all("a", recursive=False) + child = children[0] + chapterTitle = child.attrs["title"] + chapterUrl = "https://ln.hako.vn" + child.attrs["href"] + chapterHtml = get_from_url(chapterUrl) + chapterSoup = BeautifulSoup(chapterHtml, "html.parser") + chapterContent = chapterSoup.find("div", {"id": "chapter-content"}) + chapterData = "" + if chapterContent is not None: + paragraphs = chapterContent.find_all("p", id=lambda x: x and x.isdigit()) + for paragraph in paragraphs: + chapterData += paragraph.text + "\n" + write_text_to_file(chapterData, chapterTitle + ".txt") diff --git a/hako-crawler/requirements.txt b/hako-crawler/requirements.txt new file mode 100644 index 0000000..5da028b --- /dev/null +++ b/hako-crawler/requirements.txt @@ -0,0 +1,2 @@ +beautifulsoup4==4.12.3 +soupsieve==2.6