This commit is contained in:
tiennm99 committed 2024-10-11 21:03:49 +07:00
1 parent 2a6d11f617
commit 3a341a2bf6
5 files changed
+76 -1

No files matched your search

+1
View File
@@ -0,0 +1 @@
.idea
+17 -1
View File
@@ -1,2 +1,18 @@
# atnvc-crawler
Crawl data of ATNVC
Crawl data of "Anh trai nhân vật chính", a novel from https://ln.hako.vn/sang-tac/8476-kiep-nay-la-anh-trai-cua-nhan-vat-chinh
# How to run
1. Install requirements
```
pip install -r requirements.txt
```
2. Run main.py to crawl data
```
python main.py
```
__Note__: You may get `HTTP Error 429: Too Many Requests`. Then you can try again later, and skip downloaded chapters, example skip 50 first chapters like:
```
for chapter in chapters[50:]:
```
3. The data will be saved in `data` folder
+2
View File
@@ -0,0 +1,2 @@
*
!.gitignore
+54
View File
@@ -0,0 +1,54 @@
import re
from urllib.request import Request, urlopen
from bs4 import BeautifulSoup
DEBUG = False
def get_from_url(url):
request_site = Request(url, headers={"User-Agent": "Mozilla/5.0"})
webpage = urlopen(request_site).read()
return webpage.decode("utf-8")
def write_text_to_file(text, filename):
filename = re.sub(r"[^\w_. -]", "_", filename)
f = open("data/{}".format(filename), "w", encoding="utf-8")
f.write(text)
f.close()
def read_text_from_file(filename):
f = open("data/{}".format(filename), "r", encoding="utf-8")
text = f.read()
f.close()
return text
if DEBUG:
html = read_text_from_file("_.txt")
else:
html = get_from_url("https://ln.hako.vn/sang-tac/8476-kiep-nay-la-anh-trai-cua-nhan-vat-chinh")
# write_text_to_file(html, "_.txt")
soup = BeautifulSoup(html, "html.parser")
chapters = soup.find_all("div", {"class": "chapter-name"})
"""
you may get `HTTP Error 429: Too Many Requests`
you can try again later, and skip downloaded chapters
example skip 50 first chapters like: `for chapter in chapters[50:]`
"""
for chapter in chapters:
children = chapter.find_all("a", recursive=False)
child = children[0]
chapterTitle = child.attrs["title"]
chapterUrl = "https://ln.hako.vn" + child.attrs["href"]
chapterHtml = get_from_url(chapterUrl)
chapterSoup = BeautifulSoup(chapterHtml, "html.parser")
chapterContent = chapterSoup.find("div", {"id": "chapter-content"})
chapterData = ""
if chapterContent is not None:
paragraphs = chapterContent.find_all("p", id=lambda x: x and x.isdigit())
for paragraph in paragraphs:
chapterData += paragraph.text + "\n"
write_text_to_file(chapterData, chapterTitle + ".txt")
+2
View File
@@ -0,0 +1,2 @@
beautifulsoup4==4.12.3
soupsieve==2.6