mirror of
https://github.com/tiennm99/MTTools.git
synced 2026-10-05 11:00:38 +00:00
[Init]
This commit is contained in:
1 parent
2a6d11f617
commit
3a341a2bf6
5 files changed
+76
-1
No files matched your search
@@ -0,0 +1 @@
|
||||
.idea
|
||||
+17
-1
@@ -1,2 +1,18 @@
|
||||
# atnvc-crawler
|
||||
Crawl data of ATNVC
|
||||
|
||||
Crawl data of "Anh trai nhân vật chính", a novel from https://ln.hako.vn/sang-tac/8476-kiep-nay-la-anh-trai-cua-nhan-vat-chinh
|
||||
|
||||
# How to run
|
||||
1. Install requirements
|
||||
```
|
||||
pip install -r requirements.txt
|
||||
```
|
||||
2. Run main.py to crawl data
|
||||
```
|
||||
python main.py
|
||||
```
|
||||
__Note__: You may get `HTTP Error 429: Too Many Requests`. Then you can try again later, and skip downloaded chapters, example skip 50 first chapters like:
|
||||
```
|
||||
for chapter in chapters[50:]:
|
||||
```
|
||||
3. The data will be saved in `data` folder
|
||||
@@ -0,0 +1,2 @@
|
||||
*
|
||||
!.gitignore
|
||||
@@ -0,0 +1,54 @@
|
||||
import re
|
||||
from urllib.request import Request, urlopen
|
||||
from bs4 import BeautifulSoup
|
||||
|
||||
DEBUG = False
|
||||
|
||||
|
||||
def get_from_url(url):
|
||||
request_site = Request(url, headers={"User-Agent": "Mozilla/5.0"})
|
||||
webpage = urlopen(request_site).read()
|
||||
return webpage.decode("utf-8")
|
||||
|
||||
|
||||
def write_text_to_file(text, filename):
|
||||
filename = re.sub(r"[^\w_. -]", "_", filename)
|
||||
f = open("data/{}".format(filename), "w", encoding="utf-8")
|
||||
f.write(text)
|
||||
f.close()
|
||||
|
||||
|
||||
def read_text_from_file(filename):
|
||||
f = open("data/{}".format(filename), "r", encoding="utf-8")
|
||||
text = f.read()
|
||||
f.close()
|
||||
return text
|
||||
|
||||
|
||||
if DEBUG:
|
||||
html = read_text_from_file("_.txt")
|
||||
else:
|
||||
html = get_from_url("https://ln.hako.vn/sang-tac/8476-kiep-nay-la-anh-trai-cua-nhan-vat-chinh")
|
||||
# write_text_to_file(html, "_.txt")
|
||||
|
||||
soup = BeautifulSoup(html, "html.parser")
|
||||
chapters = soup.find_all("div", {"class": "chapter-name"})
|
||||
"""
|
||||
you may get `HTTP Error 429: Too Many Requests`
|
||||
you can try again later, and skip downloaded chapters
|
||||
example skip 50 first chapters like: `for chapter in chapters[50:]`
|
||||
"""
|
||||
for chapter in chapters:
|
||||
children = chapter.find_all("a", recursive=False)
|
||||
child = children[0]
|
||||
chapterTitle = child.attrs["title"]
|
||||
chapterUrl = "https://ln.hako.vn" + child.attrs["href"]
|
||||
chapterHtml = get_from_url(chapterUrl)
|
||||
chapterSoup = BeautifulSoup(chapterHtml, "html.parser")
|
||||
chapterContent = chapterSoup.find("div", {"id": "chapter-content"})
|
||||
chapterData = ""
|
||||
if chapterContent is not None:
|
||||
paragraphs = chapterContent.find_all("p", id=lambda x: x and x.isdigit())
|
||||
for paragraph in paragraphs:
|
||||
chapterData += paragraph.text + "\n"
|
||||
write_text_to_file(chapterData, chapterTitle + ".txt")
|
||||
@@ -0,0 +1,2 @@
|
||||
beautifulsoup4==4.12.3
|
||||
soupsieve==2.6
|
||||
Reference in new issue
Block a user