mirror of
https://github.com/tiennm99/MTTools.git
synced 2026-10-08 16:13:29 +00:00
feat: replace Python and JS downloaders with a Go crawler
Port both scripts into one Go program with a command per behavior: gallery walks the numbered ghibli.jp film galleries, scrape downloads every image referenced by a page. Downloads now run through a bounded worker pool, and non-200 responses are no longer written to disk as image files. Drop the orphaned package-lock.json, which pinned the deprecated request dependency and was the source of the repository's Dependabot alerts.
This commit is contained in:
1 parent
07b6b84151
commit
fb0964bb6e
13 files changed
+703
-492
No files matched your search
@@ -0,0 +1,11 @@
|
||||
# Build output
|
||||
/ghibli-gallery-crawler
|
||||
/ghibli-gallery-crawler.exe
|
||||
|
||||
# Downloaded images
|
||||
*.jpg
|
||||
*.jpeg
|
||||
*.png
|
||||
*.gif
|
||||
*.webp
|
||||
*.svg
|
||||
@@ -1,27 +1,66 @@
|
||||
# ghibli-gallery-crawler
|
||||
|
||||
Crawler for the Ghibli film gallery images (from 2020-08-31). Both Python and JavaScript variants included.
|
||||
Image downloader in Go (originally written in Python and JavaScript in 2020-08-31).
|
||||
|
||||
## Install
|
||||
|
||||
```bash
|
||||
go build -o ghibli-gallery-crawler .
|
||||
```
|
||||
|
||||
Or run without building: `go run . <command>`.
|
||||
|
||||
## Usage
|
||||
|
||||
```bash
|
||||
# Python
|
||||
python3 download_images.py
|
||||
Two commands, one per behavior carried over from the original scripts.
|
||||
|
||||
# JavaScript
|
||||
npm install
|
||||
node download_images.js
|
||||
### `gallery` — numbered ghibli.jp galleries
|
||||
|
||||
Downloads `<film>001.jpg` … `<film>050.jpg` for each film, into one directory per film:
|
||||
|
||||
```bash
|
||||
# All 8 default films into ./marnie, ./kaguyahime, ...
|
||||
ghibli-gallery-crawler gallery
|
||||
|
||||
# Pick films, image count, and output location
|
||||
ghibli-gallery-crawler gallery -films ponyo,chihiro -count 50 -out ./images
|
||||
```
|
||||
|
||||
Edit the script to set the source URLs and target directory before running.
|
||||
| Flag | Default | Meaning |
|
||||
|---|---|---|
|
||||
| `-films` | the 8 films below | comma-separated film slugs |
|
||||
| `-count` | `50` | highest image number tried per film |
|
||||
| `-out` | `.` | where the per-film directories are created |
|
||||
| `-base` | `https://www.ghibli.jp/gallery/` | base gallery URL |
|
||||
| `-concurrency` | `8` | parallel downloads |
|
||||
| `-timeout` | `30s` | per-request timeout |
|
||||
| `-user-agent` | tool identifier | `User-Agent` header to send |
|
||||
|
||||
## Input format
|
||||
Default films: `marnie`, `kaguyahime`, `kazetachinu`, `kokurikozaka`, `karigurashi`, `ponyo`, `ged`, `chihiro`.
|
||||
|
||||
Both variants read a hardcoded list of image URLs defined at the top of each script. Edit the URL array in `download_images.py` or `download_images.js` to point at your images and set the output directory path.
|
||||
Image numbers are a fixed range, so films with fewer than `-count` images are normal — those are reported as `missing`, not as failures.
|
||||
|
||||
### `scrape` — every image on a page
|
||||
|
||||
Parses a page's HTML, collects every `<img src>`, resolves relative URLs, and downloads them:
|
||||
|
||||
```bash
|
||||
# Saves into ./en.wikipedia.org (the URL's host)
|
||||
ghibli-gallery-crawler scrape https://en.wikipedia.org/wiki/Studio_Ghibli
|
||||
|
||||
# Explicit output directory
|
||||
ghibli-gallery-crawler scrape https://example.com -path ./images
|
||||
```
|
||||
|
||||
Flags: `-path` (default: the URL's host), plus `-concurrency`, `-timeout`, and `-user-agent` as above.
|
||||
|
||||
Query strings are stripped from image URLs so that names like `/hsts-pixel.gif?c=3.2.5` produce clean file names. `data:` URIs and duplicates are skipped.
|
||||
|
||||
## Output behavior
|
||||
|
||||
Images are saved into the configured output directory, named by their original filename from the URL. Existing files are overwritten. The Python variant uses `requests`; the JS variant uses `axios` + `fs.createWriteStream`.
|
||||
Images are named after the last path segment of their URL. Existing files with the same name are overwritten. Missing directories are created. Non-200 responses are never written to disk, so error pages cannot land in your output as `.jpg` files.
|
||||
|
||||
Note that `ghibli.jp` serves gallery images to any client but returns `403` for its HTML pages unless the request looks like a browser, so `gallery` works there while `scrape` does not.
|
||||
|
||||
## License
|
||||
|
||||
|
||||
@@ -0,0 +1,36 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"time"
|
||||
)
|
||||
|
||||
// defaultUserAgent identifies this tool to servers. Some sites reject requests
|
||||
// that do not send a User-Agent header at all, and an honest identifier lets
|
||||
// operators see who is fetching. Override it with -user-agent.
|
||||
const defaultUserAgent = "ghibli-gallery-crawler/1.0 (+https://github.com/tiennm99/ghibli-gallery-crawler)"
|
||||
|
||||
// userAgentTransport stamps every request with a User-Agent header.
|
||||
type userAgentTransport struct {
|
||||
userAgent string
|
||||
base http.RoundTripper
|
||||
}
|
||||
|
||||
func (t *userAgentTransport) RoundTrip(req *http.Request) (*http.Response, error) {
|
||||
// Clone before mutating: RoundTrippers must not modify the caller's request.
|
||||
clone := req.Clone(req.Context())
|
||||
clone.Header.Set("User-Agent", t.userAgent)
|
||||
return t.base.RoundTrip(clone)
|
||||
}
|
||||
|
||||
// newClient builds an HTTP client that applies the given timeout to each
|
||||
// request and identifies itself with userAgent.
|
||||
func newClient(timeout time.Duration, userAgent string) *http.Client {
|
||||
return &http.Client{
|
||||
Timeout: timeout,
|
||||
Transport: &userAgentTransport{
|
||||
userAgent: userAgent,
|
||||
base: http.DefaultTransport,
|
||||
},
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,112 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"net/url"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestExtractImageURLs(t *testing.T) {
|
||||
base, err := url.Parse("https://example.com/gallery/index.html")
|
||||
if err != nil {
|
||||
t.Fatalf("parsing base: %v", err)
|
||||
}
|
||||
|
||||
doc := `<html><body>
|
||||
<img src="photo.jpg">
|
||||
<img src="/absolute/other.png">
|
||||
<img src="https://cdn.example.net/remote.gif">
|
||||
<img src="/hsts-pixel.gif?c=3.2.5">
|
||||
<img src="photo.jpg">
|
||||
<img alt="no src attribute">
|
||||
<img src="data:image/png;base64,AAAA">
|
||||
</body></html>`
|
||||
|
||||
got, err := extractImageURLs(base, strings.NewReader(doc))
|
||||
if err != nil {
|
||||
t.Fatalf("extractImageURLs: %v", err)
|
||||
}
|
||||
|
||||
want := []string{
|
||||
"https://example.com/gallery/photo.jpg",
|
||||
"https://example.com/absolute/other.png",
|
||||
"https://cdn.example.net/remote.gif",
|
||||
"https://example.com/hsts-pixel.gif", // query stripped
|
||||
}
|
||||
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("got %d URLs %v, want %d %v", len(got), got, len(want), want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("URL %d = %q, want %q", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestGalleryJobs(t *testing.T) {
|
||||
jobs, err := galleryJobs("https://www.ghibli.jp/gallery/", []string{"ponyo", "ged"}, 3, "out")
|
||||
if err != nil {
|
||||
t.Fatalf("galleryJobs: %v", err)
|
||||
}
|
||||
|
||||
if len(jobs) != 6 {
|
||||
t.Fatalf("got %d jobs, want 6", len(jobs))
|
||||
}
|
||||
|
||||
// Numbers are zero-padded to three digits, as in the original script.
|
||||
if want := "https://www.ghibli.jp/gallery/ponyo001.jpg"; jobs[0].url != want {
|
||||
t.Errorf("first URL = %q, want %q", jobs[0].url, want)
|
||||
}
|
||||
if want := filepath.Join("out", "ponyo"); jobs[0].dir != want {
|
||||
t.Errorf("first dir = %q, want %q", jobs[0].dir, want)
|
||||
}
|
||||
if want := "https://www.ghibli.jp/gallery/ged003.jpg"; jobs[5].url != want {
|
||||
t.Errorf("last URL = %q, want %q", jobs[5].url, want)
|
||||
}
|
||||
}
|
||||
|
||||
func TestFileNameFromURL(t *testing.T) {
|
||||
cases := []struct {
|
||||
in string
|
||||
want string
|
||||
wantErr bool
|
||||
}{
|
||||
{in: "https://example.com/a/b/chihiro001.jpg", want: "chihiro001.jpg"},
|
||||
{in: "https://example.com/image.png", want: "image.png"},
|
||||
{in: "https://example.com/", wantErr: true},
|
||||
{in: "https://example.com", wantErr: true},
|
||||
}
|
||||
|
||||
for _, c := range cases {
|
||||
got, err := fileNameFromURL(c.in)
|
||||
if c.wantErr {
|
||||
if err == nil {
|
||||
t.Errorf("fileNameFromURL(%q) = %q, want error", c.in, got)
|
||||
}
|
||||
continue
|
||||
}
|
||||
if err != nil {
|
||||
t.Errorf("fileNameFromURL(%q): %v", c.in, err)
|
||||
continue
|
||||
}
|
||||
if got != c.want {
|
||||
t.Errorf("fileNameFromURL(%q) = %q, want %q", c.in, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSplitFilms(t *testing.T) {
|
||||
got := splitFilms(" ponyo , ,ged, ../escape ")
|
||||
want := []string{"ponyo", "ged"}
|
||||
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("got %v, want %v", got, want)
|
||||
}
|
||||
for i := range want {
|
||||
if got[i] != want[i] {
|
||||
t.Errorf("slug %d = %q, want %q", i, got[i], want[i])
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,157 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"os"
|
||||
"path"
|
||||
"path/filepath"
|
||||
"sync"
|
||||
)
|
||||
|
||||
// job is a single image to fetch and the directory to store it in.
|
||||
type job struct {
|
||||
url string
|
||||
dir string
|
||||
}
|
||||
|
||||
// errNotFound marks images the server does not have. The numbered gallery mode
|
||||
// always asks for a fixed range, so missing images are expected, not failures.
|
||||
var errNotFound = errors.New("not found")
|
||||
|
||||
// download fetches every job using at most concurrency parallel requests and
|
||||
// reports how many succeeded. Individual failures are logged and do not stop
|
||||
// the run.
|
||||
func download(client *http.Client, jobs []job, concurrency int) error {
|
||||
if concurrency < 1 {
|
||||
concurrency = 1
|
||||
}
|
||||
if concurrency > len(jobs) {
|
||||
concurrency = len(jobs)
|
||||
}
|
||||
|
||||
// Create directories up front so parallel workers never race on MkdirAll.
|
||||
for _, j := range jobs {
|
||||
if err := os.MkdirAll(j.dir, 0o755); err != nil {
|
||||
return fmt.Errorf("creating %s: %w", j.dir, err)
|
||||
}
|
||||
}
|
||||
|
||||
var (
|
||||
queue = make(chan job)
|
||||
wg sync.WaitGroup
|
||||
mu sync.Mutex
|
||||
saved int
|
||||
missing int
|
||||
failed int
|
||||
)
|
||||
|
||||
for range concurrency {
|
||||
wg.Add(1)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
for j := range queue {
|
||||
size, err := downloadOne(client, j)
|
||||
|
||||
mu.Lock()
|
||||
switch {
|
||||
case err == nil:
|
||||
saved++
|
||||
fmt.Printf("saved %s (%s)\n", j.url, humanSize(size))
|
||||
case errors.Is(err, errNotFound):
|
||||
missing++
|
||||
default:
|
||||
failed++
|
||||
fmt.Fprintf(os.Stderr, "failed %s: %v\n", j.url, err)
|
||||
}
|
||||
mu.Unlock()
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
for _, j := range jobs {
|
||||
queue <- j
|
||||
}
|
||||
close(queue)
|
||||
wg.Wait()
|
||||
|
||||
fmt.Printf("Done: %d saved, %d missing, %d failed\n", saved, missing, failed)
|
||||
if saved == 0 && failed > 0 {
|
||||
return fmt.Errorf("every download failed")
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// downloadOne writes a single image to disk and returns the bytes written. An
|
||||
// existing file with the same name is overwritten.
|
||||
func downloadOne(client *http.Client, j job) (int64, error) {
|
||||
name, err := fileNameFromURL(j.url)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
|
||||
resp, err := client.Get(j.url)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode == http.StatusNotFound || resp.StatusCode == http.StatusForbidden {
|
||||
return 0, errNotFound
|
||||
}
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
// Bail out before writing, so error pages never land on disk as images.
|
||||
return 0, fmt.Errorf("unexpected status %s", resp.Status)
|
||||
}
|
||||
|
||||
dest := filepath.Join(j.dir, name)
|
||||
f, err := os.Create(dest)
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
defer f.Close()
|
||||
|
||||
written, err := io.Copy(f, resp.Body)
|
||||
if err != nil {
|
||||
// Leave no truncated file behind on a mid-transfer error.
|
||||
os.Remove(dest)
|
||||
return 0, err
|
||||
}
|
||||
return written, nil
|
||||
}
|
||||
|
||||
// fileNameFromURL derives a safe local file name from the last path segment of
|
||||
// an image URL.
|
||||
func fileNameFromURL(rawURL string) (string, error) {
|
||||
parsed, err := url.Parse(rawURL)
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
name := path.Base(parsed.Path)
|
||||
// path.Base returns "." or "/" when there is nothing usable to take.
|
||||
if name == "" || name == "." || name == "/" || name == ".." {
|
||||
return "", fmt.Errorf("cannot derive a file name from %q", rawURL)
|
||||
}
|
||||
// Guard against a remote path smuggling in separators.
|
||||
return filepath.Base(name), nil
|
||||
}
|
||||
|
||||
// humanSize formats a byte count for display.
|
||||
func humanSize(n int64) string {
|
||||
const unit = 1024
|
||||
if n < unit {
|
||||
return fmt.Sprintf("%d B", n)
|
||||
}
|
||||
value := float64(n)
|
||||
for _, suffix := range []string{"KiB", "MiB", "GiB"} {
|
||||
value /= unit
|
||||
if value < unit {
|
||||
return fmt.Sprintf("%.1f %s", value, suffix)
|
||||
}
|
||||
}
|
||||
return fmt.Sprintf("%.1f TiB", value)
|
||||
}
|
||||
@@ -1,54 +0,0 @@
|
||||
const fs = require('fs')
|
||||
const request = require('request')
|
||||
const {
|
||||
cookieCompare
|
||||
} = require('tough-cookie')
|
||||
|
||||
const download = (url, path, callback) => {
|
||||
request.head(url, (err, res, body) => {
|
||||
request(url)
|
||||
.pipe(fs.createWriteStream(path))
|
||||
.on('close', callback)
|
||||
})
|
||||
}
|
||||
|
||||
// const url = 'https://…'
|
||||
// const path = './images/image.png'
|
||||
|
||||
// download(url, path, () => {
|
||||
// console.log('✅ Done!')
|
||||
// })
|
||||
|
||||
// let urlConst = "https://www.iamag.co/wp-content/uploads/2017/07/yourname-background"
|
||||
// for (let i = 1; i < 200; i++) {
|
||||
// download(urlConst + i + ".jpeg", "img/" + i + ".jpeg", () => {
|
||||
// console.log("Done: " + i);
|
||||
// })
|
||||
// }
|
||||
|
||||
function lpad(n, width, z) {
|
||||
z = z || '0';
|
||||
n = n + '';
|
||||
return n.length >= width ? n : new Array(width - n.length + 1).join(z) + n;
|
||||
}
|
||||
|
||||
let ghibli = "http://www.ghibli.jp/gallery/";
|
||||
let listFilm = ["marnie", "kaguyahime", "kazetachinu", "kokurikozaka", "karigurashi", "ponyo", "ged", "chihiro"];
|
||||
listFilm.forEach(film => {
|
||||
let filmDir = "./" + film;
|
||||
fs.mkdir(filmDir, function(err) {
|
||||
if (err) {
|
||||
console.log(err);
|
||||
} else {
|
||||
console.log("Created " + filmDir);
|
||||
}
|
||||
});
|
||||
for (let i = 1; i <= 50; i++) {
|
||||
let imgName = lpad(i, 3) + ".jpg";
|
||||
let imgUrl = ghibli + film + imgName;
|
||||
console.log(imgUrl);
|
||||
download(imgUrl, filmDir + "/" + imgName, () => {
|
||||
console.log("Done: " + imgName);
|
||||
})
|
||||
}
|
||||
})
|
||||
@@ -1,90 +0,0 @@
|
||||
import requests
|
||||
import os
|
||||
from tqdm import tqdm
|
||||
from bs4 import BeautifulSoup as bs
|
||||
from urllib.parse import urljoin, urlparse
|
||||
|
||||
|
||||
def is_valid(url):
|
||||
"""
|
||||
Checks whether `url` is a valid URL.
|
||||
"""
|
||||
parsed = urlparse(url)
|
||||
return bool(parsed.netloc) and bool(parsed.scheme)
|
||||
|
||||
|
||||
def get_all_images(url):
|
||||
"""
|
||||
Returns all image URLs on a single `url`
|
||||
"""
|
||||
soup = bs(requests.get(url).content, "html.parser")
|
||||
urls = []
|
||||
for img in tqdm(soup.find_all("img"), "Extracting images"):
|
||||
img_url = img.attrs.get("src")
|
||||
if not img_url:
|
||||
# if img does not contain src attribute, just skip
|
||||
continue
|
||||
# make the URL absolute by joining domain with the URL that is just extracted
|
||||
img_url = urljoin(url, img_url)
|
||||
# remove URLs like '/hsts-pixel.gif?c=3.2.5'
|
||||
try:
|
||||
pos = img_url.index("?")
|
||||
img_url = img_url[:pos]
|
||||
except ValueError:
|
||||
pass
|
||||
# finally, if the url is valid
|
||||
if is_valid(img_url):
|
||||
urls.append(img_url)
|
||||
return urls
|
||||
|
||||
|
||||
def download(url, pathname):
|
||||
"""
|
||||
Downloads a file given an URL and puts it in the folder `pathname`
|
||||
"""
|
||||
# if path doesn't exist, make that path dir
|
||||
if not os.path.isdir(pathname):
|
||||
os.makedirs(pathname)
|
||||
# download the body of response by chunk, not immediately
|
||||
response = requests.get(url, stream=True)
|
||||
|
||||
# get the total file size
|
||||
file_size = int(response.headers.get("Content-Length", 0))
|
||||
|
||||
# get the file name
|
||||
filename = os.path.join(pathname, url.split("/")[-1])
|
||||
|
||||
# progress bar, changing the unit to bytes instead of iteration (default by tqdm)
|
||||
progress = tqdm(response.iter_content(1024), f"Downloading {filename}", total=file_size, unit="B", unit_scale=True, unit_divisor=1024)
|
||||
with open(filename, "wb") as f:
|
||||
for data in progress:
|
||||
# write data read to the file
|
||||
f.write(data)
|
||||
# update the progress bar manually
|
||||
progress.update(len(data))
|
||||
|
||||
|
||||
def main(url, path):
|
||||
# get all images
|
||||
imgs = get_all_images(url)
|
||||
for img in imgs:
|
||||
# for each img, download it
|
||||
download(img, path)
|
||||
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import argparse
|
||||
parser = argparse.ArgumentParser(description="This script downloads all images from a web page")
|
||||
parser.add_argument("url", help="The URL of the web page you want to download images")
|
||||
parser.add_argument("-p", "--path", help="The Directory you want to store your images, default is the domain of URL passed")
|
||||
|
||||
args = parser.parse_args()
|
||||
url = args.url
|
||||
path = args.path
|
||||
|
||||
if not path:
|
||||
# if path isn't specified, use the domain name of that url as the folder name
|
||||
path = urlparse(url).netloc
|
||||
|
||||
main(url, path)
|
||||
@@ -0,0 +1,34 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"net/url"
|
||||
"path/filepath"
|
||||
)
|
||||
|
||||
// galleryJobs builds the download list for the numbered ghibli.jp galleries:
|
||||
// for each film, images are named <film>001.jpg through <film><count>.jpg and
|
||||
// are stored in a directory named after the film.
|
||||
func galleryJobs(base string, films []string, count int, out string) ([]job, error) {
|
||||
baseURL, err := url.Parse(base)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parsing base URL %q: %w", base, err)
|
||||
}
|
||||
if baseURL.Scheme == "" || baseURL.Host == "" {
|
||||
return nil, fmt.Errorf("base URL %q must be absolute", base)
|
||||
}
|
||||
|
||||
jobs := make([]job, 0, len(films)*count)
|
||||
for _, film := range films {
|
||||
dir := filepath.Join(out, film)
|
||||
for i := 1; i <= count; i++ {
|
||||
name := fmt.Sprintf("%s%03d.jpg", film, i)
|
||||
imageURL, err := url.JoinPath(base, name)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("building URL for %s: %w", name, err)
|
||||
}
|
||||
jobs = append(jobs, job{url: imageURL, dir: dir})
|
||||
}
|
||||
}
|
||||
return jobs, nil
|
||||
}
|
||||
@@ -0,0 +1,5 @@
|
||||
module github.com/tiennm99/ghibli-gallery-crawler
|
||||
|
||||
go 1.26.4
|
||||
|
||||
require golang.org/x/net v0.57.0
|
||||
@@ -0,0 +1,2 @@
|
||||
golang.org/x/net v0.57.0 h1:K5+3DljvIuDG9/Jv9rvyMywYNFCQ9RSUY6OOTTkT+tE=
|
||||
golang.org/x/net v0.57.0/go.mod h1:KpXc8iv+r3XplLAG/f7Jsf9RPszJzdR0f58q9vGOuEU=
|
||||
@@ -0,0 +1,204 @@
|
||||
// Command ghibli-gallery-crawler downloads images from web pages.
|
||||
//
|
||||
// It offers two modes, mirroring the two scripts it replaces:
|
||||
//
|
||||
// scrape - parse a page's <img> tags and download every image found
|
||||
// gallery - walk the numbered per-film galleries on ghibli.jp
|
||||
package main
|
||||
|
||||
import (
|
||||
"flag"
|
||||
"fmt"
|
||||
"net/url"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
const (
|
||||
// defaultGalleryBase is where the numbered film galleries live. The
|
||||
// original script used http://; https:// reaches the same files.
|
||||
defaultGalleryBase = "https://www.ghibli.jp/gallery/"
|
||||
// defaultGalleryCount is the highest image number tried per film.
|
||||
defaultGalleryCount = 50
|
||||
)
|
||||
|
||||
// defaultFilms lists the gallery slugs the original JavaScript script crawled.
|
||||
var defaultFilms = []string{
|
||||
"marnie", "kaguyahime", "kazetachinu", "kokurikozaka",
|
||||
"karigurashi", "ponyo", "ged", "chihiro",
|
||||
}
|
||||
|
||||
func main() {
|
||||
if err := run(os.Args[1:]); err != nil {
|
||||
fmt.Fprintln(os.Stderr, "error:", err)
|
||||
os.Exit(1)
|
||||
}
|
||||
}
|
||||
|
||||
func run(args []string) error {
|
||||
if len(args) == 0 {
|
||||
usage()
|
||||
return fmt.Errorf("no command given")
|
||||
}
|
||||
|
||||
switch args[0] {
|
||||
case "scrape":
|
||||
return runScrape(args[1:])
|
||||
case "gallery":
|
||||
return runGallery(args[1:])
|
||||
case "help", "-h", "--help":
|
||||
usage()
|
||||
return nil
|
||||
default:
|
||||
usage()
|
||||
return fmt.Errorf("unknown command %q", args[0])
|
||||
}
|
||||
}
|
||||
|
||||
func usage() {
|
||||
fmt.Fprint(os.Stderr, `ghibli-gallery-crawler downloads images from web pages.
|
||||
|
||||
Usage:
|
||||
ghibli-gallery-crawler scrape <url> [flags]
|
||||
ghibli-gallery-crawler gallery [flags]
|
||||
|
||||
Commands:
|
||||
scrape Download every <img> found on a single page.
|
||||
gallery Download the numbered ghibli.jp galleries for a list of films.
|
||||
|
||||
Run a command with -h to see its flags.
|
||||
`)
|
||||
}
|
||||
|
||||
// runScrape downloads every image referenced by a single page. The default
|
||||
// output directory is the page's host, matching the Python script it replaces.
|
||||
func runScrape(args []string) error {
|
||||
fs := flag.NewFlagSet("scrape", flag.ExitOnError)
|
||||
path := fs.String("path", "", "directory to store images in (default: the URL's host)")
|
||||
concurrency := fs.Int("concurrency", 8, "number of parallel downloads")
|
||||
timeout := fs.Duration("timeout", 30*time.Second, "per-request timeout")
|
||||
userAgent := fs.String("user-agent", defaultUserAgent, "User-Agent header to send")
|
||||
fs.Usage = func() {
|
||||
fmt.Fprintln(fs.Output(), "Usage: ghibli-gallery-crawler scrape <url> [flags]")
|
||||
fs.PrintDefaults()
|
||||
}
|
||||
positional, err := parseWithPositionals(fs, args)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if len(positional) != 1 {
|
||||
fs.Usage()
|
||||
return fmt.Errorf("scrape needs exactly one URL argument, got %d", len(positional))
|
||||
}
|
||||
pageURL := positional[0]
|
||||
|
||||
parsed, err := url.Parse(pageURL)
|
||||
if err != nil {
|
||||
return fmt.Errorf("parsing %q: %w", pageURL, err)
|
||||
}
|
||||
if parsed.Scheme == "" || parsed.Host == "" {
|
||||
return fmt.Errorf("%q is not an absolute URL", pageURL)
|
||||
}
|
||||
|
||||
dir := *path
|
||||
if dir == "" {
|
||||
dir = parsed.Host
|
||||
}
|
||||
|
||||
client := newClient(*timeout, *userAgent)
|
||||
|
||||
imageURLs, err := scrapeImageURLs(client, parsed)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if len(imageURLs) == 0 {
|
||||
fmt.Printf("No images found on %s\n", pageURL)
|
||||
return nil
|
||||
}
|
||||
fmt.Printf("Found %d image(s) on %s\n", len(imageURLs), pageURL)
|
||||
|
||||
jobs := make([]job, 0, len(imageURLs))
|
||||
for _, imageURL := range imageURLs {
|
||||
jobs = append(jobs, job{url: imageURL, dir: dir})
|
||||
}
|
||||
return download(client, jobs, *concurrency)
|
||||
}
|
||||
|
||||
// runGallery downloads images named <film>001.jpg .. <film>NNN.jpg for each
|
||||
// film, storing each film's images in its own directory. This mirrors the
|
||||
// JavaScript script it replaces.
|
||||
func runGallery(args []string) error {
|
||||
fs := flag.NewFlagSet("gallery", flag.ExitOnError)
|
||||
base := fs.String("base", defaultGalleryBase, "base gallery URL")
|
||||
films := fs.String("films", strings.Join(defaultFilms, ","), "comma-separated film slugs")
|
||||
count := fs.Int("count", defaultGalleryCount, "highest image number to try per film")
|
||||
out := fs.String("out", ".", "directory to create the per-film directories in")
|
||||
concurrency := fs.Int("concurrency", 8, "number of parallel downloads")
|
||||
timeout := fs.Duration("timeout", 30*time.Second, "per-request timeout")
|
||||
userAgent := fs.String("user-agent", defaultUserAgent, "User-Agent header to send")
|
||||
fs.Usage = func() {
|
||||
fmt.Fprintln(fs.Output(), "Usage: ghibli-gallery-crawler gallery [flags]")
|
||||
fs.PrintDefaults()
|
||||
}
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
if *count < 1 {
|
||||
return fmt.Errorf("-count must be at least 1, got %d", *count)
|
||||
}
|
||||
|
||||
slugs := splitFilms(*films)
|
||||
if len(slugs) == 0 {
|
||||
return fmt.Errorf("-films must name at least one film")
|
||||
}
|
||||
|
||||
jobs, err := galleryJobs(*base, slugs, *count, *out)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
fmt.Printf("Trying %d image(s) across %d film(s)\n", len(jobs), len(slugs))
|
||||
|
||||
client := newClient(*timeout, *userAgent)
|
||||
return download(client, jobs, *concurrency)
|
||||
}
|
||||
|
||||
// parseWithPositionals parses flags that may appear before or after positional
|
||||
// arguments. The standard flag package stops at the first non-flag argument, so
|
||||
// the remainder is fed back through the parser one positional at a time.
|
||||
func parseWithPositionals(fs *flag.FlagSet, args []string) ([]string, error) {
|
||||
var positional []string
|
||||
for len(args) > 0 {
|
||||
if err := fs.Parse(args); err != nil {
|
||||
return nil, err
|
||||
}
|
||||
args = fs.Args()
|
||||
if len(args) > 0 {
|
||||
positional = append(positional, args[0])
|
||||
args = args[1:]
|
||||
}
|
||||
}
|
||||
return positional, nil
|
||||
}
|
||||
|
||||
// splitFilms turns a comma-separated flag value into clean slugs, dropping
|
||||
// empty entries and anything that would escape the output directory.
|
||||
func splitFilms(value string) []string {
|
||||
var slugs []string
|
||||
for _, raw := range strings.Split(value, ",") {
|
||||
slug := strings.TrimSpace(raw)
|
||||
if slug == "" {
|
||||
continue
|
||||
}
|
||||
// A slug becomes a directory name, so reject path separators.
|
||||
if slug != filepath.Base(slug) || slug == "." || slug == ".." {
|
||||
fmt.Fprintf(os.Stderr, "skipping invalid film slug %q\n", slug)
|
||||
continue
|
||||
}
|
||||
slugs = append(slugs, slug)
|
||||
}
|
||||
return slugs
|
||||
}
|
||||
Generated
-337
@@ -1,337 +0,0 @@
|
||||
{
|
||||
"requires": true,
|
||||
"lockfileVersion": 1,
|
||||
"dependencies": {
|
||||
"ajv": {
|
||||
"version": "6.12.4",
|
||||
"resolved": "https://registry.npmjs.org/ajv/-/ajv-6.12.4.tgz",
|
||||
"integrity": "sha512-eienB2c9qVQs2KWexhkrdMLVDoIQCz5KSeLxwg9Lzk4DOfBtIK9PQwwufcsn1jjGuf9WZmqPMbGxOzfcuphJCQ==",
|
||||
"requires": {
|
||||
"fast-deep-equal": "^3.1.1",
|
||||
"fast-json-stable-stringify": "^2.0.0",
|
||||
"json-schema-traverse": "^0.4.1",
|
||||
"uri-js": "^4.2.2"
|
||||
}
|
||||
},
|
||||
"asn1": {
|
||||
"version": "0.2.4",
|
||||
"resolved": "https://registry.npmjs.org/asn1/-/asn1-0.2.4.tgz",
|
||||
"integrity": "sha512-jxwzQpLQjSmWXgwaCZE9Nz+glAG01yF1QnWgbhGwHI5A6FRIEY6IVqtHhIepHqI7/kyEyQEagBC5mBEFlIYvdg==",
|
||||
"requires": {
|
||||
"safer-buffer": "~2.1.0"
|
||||
}
|
||||
},
|
||||
"assert-plus": {
|
||||
"version": "1.0.0",
|
||||
"resolved": "https://registry.npmjs.org/assert-plus/-/assert-plus-1.0.0.tgz",
|
||||
"integrity": "sha1-8S4PPF13sLHN2RRpQuTpbB5N1SU="
|
||||
},
|
||||
"asynckit": {
|
||||
"version": "0.4.0",
|
||||
"resolved": "https://registry.npmjs.org/asynckit/-/asynckit-0.4.0.tgz",
|
||||
"integrity": "sha1-x57Zf380y48robyXkLzDZkdLS3k="
|
||||
},
|
||||
"aws-sign2": {
|
||||
"version": "0.7.0",
|
||||
"resolved": "https://registry.npmjs.org/aws-sign2/-/aws-sign2-0.7.0.tgz",
|
||||
"integrity": "sha1-tG6JCTSpWR8tL2+G1+ap8bP+dqg="
|
||||
},
|
||||
"aws4": {
|
||||
"version": "1.10.1",
|
||||
"resolved": "https://registry.npmjs.org/aws4/-/aws4-1.10.1.tgz",
|
||||
"integrity": "sha512-zg7Hz2k5lI8kb7U32998pRRFin7zJlkfezGJjUc2heaD4Pw2wObakCDVzkKztTm/Ln7eiVvYsjqak0Ed4LkMDA=="
|
||||
},
|
||||
"bcrypt-pbkdf": {
|
||||
"version": "1.0.2",
|
||||
"resolved": "https://registry.npmjs.org/bcrypt-pbkdf/-/bcrypt-pbkdf-1.0.2.tgz",
|
||||
"integrity": "sha1-pDAdOJtqQ/m2f/PKEaP2Y342Dp4=",
|
||||
"requires": {
|
||||
"tweetnacl": "^0.14.3"
|
||||
}
|
||||
},
|
||||
"caseless": {
|
||||
"version": "0.12.0",
|
||||
"resolved": "https://registry.npmjs.org/caseless/-/caseless-0.12.0.tgz",
|
||||
"integrity": "sha1-G2gcIf+EAzyCZUMJBolCDRhxUdw="
|
||||
},
|
||||
"combined-stream": {
|
||||
"version": "1.0.8",
|
||||
"resolved": "https://registry.npmjs.org/combined-stream/-/combined-stream-1.0.8.tgz",
|
||||
"integrity": "sha512-FQN4MRfuJeHf7cBbBMJFXhKSDq+2kAArBlmRBvcvFE5BB1HZKXtSFASDhdlz9zOYwxh8lDdnvmMOe/+5cdoEdg==",
|
||||
"requires": {
|
||||
"delayed-stream": "~1.0.0"
|
||||
}
|
||||
},
|
||||
"core-util-is": {
|
||||
"version": "1.0.2",
|
||||
"resolved": "https://registry.npmjs.org/core-util-is/-/core-util-is-1.0.2.tgz",
|
||||
"integrity": "sha1-tf1UIgqivFq1eqtxQMlAdUUDwac="
|
||||
},
|
||||
"dashdash": {
|
||||
"version": "1.14.1",
|
||||
"resolved": "https://registry.npmjs.org/dashdash/-/dashdash-1.14.1.tgz",
|
||||
"integrity": "sha1-hTz6D3y+L+1d4gMmuN1YEDX24vA=",
|
||||
"requires": {
|
||||
"assert-plus": "^1.0.0"
|
||||
}
|
||||
},
|
||||
"delayed-stream": {
|
||||
"version": "1.0.0",
|
||||
"resolved": "https://registry.npmjs.org/delayed-stream/-/delayed-stream-1.0.0.tgz",
|
||||
"integrity": "sha1-3zrhmayt+31ECqrgsp4icrJOxhk="
|
||||
},
|
||||
"ecc-jsbn": {
|
||||
"version": "0.1.2",
|
||||
"resolved": "https://registry.npmjs.org/ecc-jsbn/-/ecc-jsbn-0.1.2.tgz",
|
||||
"integrity": "sha1-OoOpBOVDUyh4dMVkt1SThoSamMk=",
|
||||
"requires": {
|
||||
"jsbn": "~0.1.0",
|
||||
"safer-buffer": "^2.1.0"
|
||||
}
|
||||
},
|
||||
"extend": {
|
||||
"version": "3.0.2",
|
||||
"resolved": "https://registry.npmjs.org/extend/-/extend-3.0.2.tgz",
|
||||
"integrity": "sha512-fjquC59cD7CyW6urNXK0FBufkZcoiGG80wTuPujX590cB5Ttln20E2UB4S/WARVqhXffZl2LNgS+gQdPIIim/g=="
|
||||
},
|
||||
"extsprintf": {
|
||||
"version": "1.3.0",
|
||||
"resolved": "https://registry.npmjs.org/extsprintf/-/extsprintf-1.3.0.tgz",
|
||||
"integrity": "sha1-lpGEQOMEGnpBT4xS48V06zw+HgU="
|
||||
},
|
||||
"fast-deep-equal": {
|
||||
"version": "3.1.3",
|
||||
"resolved": "https://registry.npmjs.org/fast-deep-equal/-/fast-deep-equal-3.1.3.tgz",
|
||||
"integrity": "sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q=="
|
||||
},
|
||||
"fast-json-stable-stringify": {
|
||||
"version": "2.1.0",
|
||||
"resolved": "https://registry.npmjs.org/fast-json-stable-stringify/-/fast-json-stable-stringify-2.1.0.tgz",
|
||||
"integrity": "sha512-lhd/wF+Lk98HZoTCtlVraHtfh5XYijIjalXck7saUtuanSDyLMxnHhSXEDJqHxD7msR8D0uCmqlkwjCV8xvwHw=="
|
||||
},
|
||||
"forever-agent": {
|
||||
"version": "0.6.1",
|
||||
"resolved": "https://registry.npmjs.org/forever-agent/-/forever-agent-0.6.1.tgz",
|
||||
"integrity": "sha1-+8cfDEGt6zf5bFd60e1C2P2sypE="
|
||||
},
|
||||
"form-data": {
|
||||
"version": "2.3.3",
|
||||
"resolved": "https://registry.npmjs.org/form-data/-/form-data-2.3.3.tgz",
|
||||
"integrity": "sha512-1lLKB2Mu3aGP1Q/2eCOx0fNbRMe7XdwktwOruhfqqd0rIJWwN4Dh+E3hrPSlDCXnSR7UtZ1N38rVXm+6+MEhJQ==",
|
||||
"requires": {
|
||||
"asynckit": "^0.4.0",
|
||||
"combined-stream": "^1.0.6",
|
||||
"mime-types": "^2.1.12"
|
||||
}
|
||||
},
|
||||
"getpass": {
|
||||
"version": "0.1.7",
|
||||
"resolved": "https://registry.npmjs.org/getpass/-/getpass-0.1.7.tgz",
|
||||
"integrity": "sha1-Xv+OPmhNVprkyysSgmBOi6YhSfo=",
|
||||
"requires": {
|
||||
"assert-plus": "^1.0.0"
|
||||
}
|
||||
},
|
||||
"har-schema": {
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/har-schema/-/har-schema-2.0.0.tgz",
|
||||
"integrity": "sha1-qUwiJOvKwEeCoNkDVSHyRzW37JI="
|
||||
},
|
||||
"har-validator": {
|
||||
"version": "5.1.5",
|
||||
"resolved": "https://registry.npmjs.org/har-validator/-/har-validator-5.1.5.tgz",
|
||||
"integrity": "sha512-nmT2T0lljbxdQZfspsno9hgrG3Uir6Ks5afism62poxqBM6sDnMEuPmzTq8XN0OEwqKLLdh1jQI3qyE66Nzb3w==",
|
||||
"requires": {
|
||||
"ajv": "^6.12.3",
|
||||
"har-schema": "^2.0.0"
|
||||
}
|
||||
},
|
||||
"http-signature": {
|
||||
"version": "1.2.0",
|
||||
"resolved": "https://registry.npmjs.org/http-signature/-/http-signature-1.2.0.tgz",
|
||||
"integrity": "sha1-muzZJRFHcvPZW2WmCruPfBj7rOE=",
|
||||
"requires": {
|
||||
"assert-plus": "^1.0.0",
|
||||
"jsprim": "^1.2.2",
|
||||
"sshpk": "^1.7.0"
|
||||
}
|
||||
},
|
||||
"is-typedarray": {
|
||||
"version": "1.0.0",
|
||||
"resolved": "https://registry.npmjs.org/is-typedarray/-/is-typedarray-1.0.0.tgz",
|
||||
"integrity": "sha1-5HnICFjfDBsR3dppQPlgEfzaSpo="
|
||||
},
|
||||
"isstream": {
|
||||
"version": "0.1.2",
|
||||
"resolved": "https://registry.npmjs.org/isstream/-/isstream-0.1.2.tgz",
|
||||
"integrity": "sha1-R+Y/evVa+m+S4VAOaQ64uFKcCZo="
|
||||
},
|
||||
"jsbn": {
|
||||
"version": "0.1.1",
|
||||
"resolved": "https://registry.npmjs.org/jsbn/-/jsbn-0.1.1.tgz",
|
||||
"integrity": "sha1-peZUwuWi3rXyAdls77yoDA7y9RM="
|
||||
},
|
||||
"json-schema": {
|
||||
"version": "0.2.3",
|
||||
"resolved": "https://registry.npmjs.org/json-schema/-/json-schema-0.2.3.tgz",
|
||||
"integrity": "sha1-tIDIkuWaLwWVTOcnvT8qTogvnhM="
|
||||
},
|
||||
"json-schema-traverse": {
|
||||
"version": "0.4.1",
|
||||
"resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-0.4.1.tgz",
|
||||
"integrity": "sha512-xbbCH5dCYU5T8LcEhhuh7HJ88HXuW3qsI3Y0zOZFKfZEHcpWiHU/Jxzk629Brsab/mMiHQti9wMP+845RPe3Vg=="
|
||||
},
|
||||
"json-stringify-safe": {
|
||||
"version": "5.0.1",
|
||||
"resolved": "https://registry.npmjs.org/json-stringify-safe/-/json-stringify-safe-5.0.1.tgz",
|
||||
"integrity": "sha1-Epai1Y/UXxmg9s4B1lcB4sc1tus="
|
||||
},
|
||||
"jsprim": {
|
||||
"version": "1.4.1",
|
||||
"resolved": "https://registry.npmjs.org/jsprim/-/jsprim-1.4.1.tgz",
|
||||
"integrity": "sha1-MT5mvB5cwG5Di8G3SZwuXFastqI=",
|
||||
"requires": {
|
||||
"assert-plus": "1.0.0",
|
||||
"extsprintf": "1.3.0",
|
||||
"json-schema": "0.2.3",
|
||||
"verror": "1.10.0"
|
||||
}
|
||||
},
|
||||
"mime-db": {
|
||||
"version": "1.44.0",
|
||||
"resolved": "https://registry.npmjs.org/mime-db/-/mime-db-1.44.0.tgz",
|
||||
"integrity": "sha512-/NOTfLrsPBVeH7YtFPgsVWveuL+4SjjYxaQ1xtM1KMFj7HdxlBlxeyNLzhyJVx7r4rZGJAZ/6lkKCitSc/Nmpg=="
|
||||
},
|
||||
"mime-types": {
|
||||
"version": "2.1.27",
|
||||
"resolved": "https://registry.npmjs.org/mime-types/-/mime-types-2.1.27.tgz",
|
||||
"integrity": "sha512-JIhqnCasI9yD+SsmkquHBxTSEuZdQX5BuQnS2Vc7puQQQ+8yiP5AY5uWhpdv4YL4VM5c6iliiYWPgJ/nJQLp7w==",
|
||||
"requires": {
|
||||
"mime-db": "1.44.0"
|
||||
}
|
||||
},
|
||||
"oauth-sign": {
|
||||
"version": "0.9.0",
|
||||
"resolved": "https://registry.npmjs.org/oauth-sign/-/oauth-sign-0.9.0.tgz",
|
||||
"integrity": "sha512-fexhUFFPTGV8ybAtSIGbV6gOkSv8UtRbDBnAyLQw4QPKkgNlsH2ByPGtMUqdWkos6YCRmAqViwgZrJc/mRDzZQ=="
|
||||
},
|
||||
"performance-now": {
|
||||
"version": "2.1.0",
|
||||
"resolved": "https://registry.npmjs.org/performance-now/-/performance-now-2.1.0.tgz",
|
||||
"integrity": "sha1-Ywn04OX6kT7BxpMHrjZLSzd8nns="
|
||||
},
|
||||
"psl": {
|
||||
"version": "1.8.0",
|
||||
"resolved": "https://registry.npmjs.org/psl/-/psl-1.8.0.tgz",
|
||||
"integrity": "sha512-RIdOzyoavK+hA18OGGWDqUTsCLhtA7IcZ/6NCs4fFJaHBDab+pDDmDIByWFRQJq2Cd7r1OoQxBGKOaztq+hjIQ=="
|
||||
},
|
||||
"punycode": {
|
||||
"version": "2.1.1",
|
||||
"resolved": "https://registry.npmjs.org/punycode/-/punycode-2.1.1.tgz",
|
||||
"integrity": "sha512-XRsRjdf+j5ml+y/6GKHPZbrF/8p2Yga0JPtdqTIY2Xe5ohJPD9saDJJLPvp9+NSBprVvevdXZybnj2cv8OEd0A=="
|
||||
},
|
||||
"qs": {
|
||||
"version": "6.5.2",
|
||||
"resolved": "https://registry.npmjs.org/qs/-/qs-6.5.2.tgz",
|
||||
"integrity": "sha512-N5ZAX4/LxJmF+7wN74pUD6qAh9/wnvdQcjq9TZjevvXzSUo7bfmw91saqMjzGS2xq91/odN2dW/WOl7qQHNDGA=="
|
||||
},
|
||||
"request": {
|
||||
"version": "2.88.2",
|
||||
"resolved": "https://registry.npmjs.org/request/-/request-2.88.2.tgz",
|
||||
"integrity": "sha512-MsvtOrfG9ZcrOwAW+Qi+F6HbD0CWXEh9ou77uOb7FM2WPhwT7smM833PzanhJLsgXjN89Ir6V2PczXNnMpwKhw==",
|
||||
"requires": {
|
||||
"aws-sign2": "~0.7.0",
|
||||
"aws4": "^1.8.0",
|
||||
"caseless": "~0.12.0",
|
||||
"combined-stream": "~1.0.6",
|
||||
"extend": "~3.0.2",
|
||||
"forever-agent": "~0.6.1",
|
||||
"form-data": "~2.3.2",
|
||||
"har-validator": "~5.1.3",
|
||||
"http-signature": "~1.2.0",
|
||||
"is-typedarray": "~1.0.0",
|
||||
"isstream": "~0.1.2",
|
||||
"json-stringify-safe": "~5.0.1",
|
||||
"mime-types": "~2.1.19",
|
||||
"oauth-sign": "~0.9.0",
|
||||
"performance-now": "^2.1.0",
|
||||
"qs": "~6.5.2",
|
||||
"safe-buffer": "^5.1.2",
|
||||
"tough-cookie": "~2.5.0",
|
||||
"tunnel-agent": "^0.6.0",
|
||||
"uuid": "^3.3.2"
|
||||
}
|
||||
},
|
||||
"safe-buffer": {
|
||||
"version": "5.2.1",
|
||||
"resolved": "https://registry.npmjs.org/safe-buffer/-/safe-buffer-5.2.1.tgz",
|
||||
"integrity": "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="
|
||||
},
|
||||
"safer-buffer": {
|
||||
"version": "2.1.2",
|
||||
"resolved": "https://registry.npmjs.org/safer-buffer/-/safer-buffer-2.1.2.tgz",
|
||||
"integrity": "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg=="
|
||||
},
|
||||
"sshpk": {
|
||||
"version": "1.16.1",
|
||||
"resolved": "https://registry.npmjs.org/sshpk/-/sshpk-1.16.1.tgz",
|
||||
"integrity": "sha512-HXXqVUq7+pcKeLqqZj6mHFUMvXtOJt1uoUx09pFW6011inTMxqI8BA8PM95myrIyyKwdnzjdFjLiE6KBPVtJIg==",
|
||||
"requires": {
|
||||
"asn1": "~0.2.3",
|
||||
"assert-plus": "^1.0.0",
|
||||
"bcrypt-pbkdf": "^1.0.0",
|
||||
"dashdash": "^1.12.0",
|
||||
"ecc-jsbn": "~0.1.1",
|
||||
"getpass": "^0.1.1",
|
||||
"jsbn": "~0.1.0",
|
||||
"safer-buffer": "^2.0.2",
|
||||
"tweetnacl": "~0.14.0"
|
||||
}
|
||||
},
|
||||
"tough-cookie": {
|
||||
"version": "2.5.0",
|
||||
"resolved": "https://registry.npmjs.org/tough-cookie/-/tough-cookie-2.5.0.tgz",
|
||||
"integrity": "sha512-nlLsUzgm1kfLXSXfRZMc1KLAugd4hqJHDTvc2hDIwS3mZAfMEuMbc03SujMF+GEcpaX/qboeycw6iO8JwVv2+g==",
|
||||
"requires": {
|
||||
"psl": "^1.1.28",
|
||||
"punycode": "^2.1.1"
|
||||
}
|
||||
},
|
||||
"tunnel-agent": {
|
||||
"version": "0.6.0",
|
||||
"resolved": "https://registry.npmjs.org/tunnel-agent/-/tunnel-agent-0.6.0.tgz",
|
||||
"integrity": "sha1-J6XeoGs2sEoKmWZ3SykIaPD8QP0=",
|
||||
"requires": {
|
||||
"safe-buffer": "^5.0.1"
|
||||
}
|
||||
},
|
||||
"tweetnacl": {
|
||||
"version": "0.14.5",
|
||||
"resolved": "https://registry.npmjs.org/tweetnacl/-/tweetnacl-0.14.5.tgz",
|
||||
"integrity": "sha1-WuaBd/GS1EViadEIr6k/+HQ/T2Q="
|
||||
},
|
||||
"uri-js": {
|
||||
"version": "4.4.0",
|
||||
"resolved": "https://registry.npmjs.org/uri-js/-/uri-js-4.4.0.tgz",
|
||||
"integrity": "sha512-B0yRTzYdUCCn9n+F4+Gh4yIDtMQcaJsmYBDsTSG8g/OejKBodLQ2IHfN3bM7jUsRXndopT7OIXWdYqc1fjmV6g==",
|
||||
"requires": {
|
||||
"punycode": "^2.1.0"
|
||||
}
|
||||
},
|
||||
"uuid": {
|
||||
"version": "3.4.0",
|
||||
"resolved": "https://registry.npmjs.org/uuid/-/uuid-3.4.0.tgz",
|
||||
"integrity": "sha512-HjSDRw6gZE5JMggctHBcjVak08+KEVhSIiDzFnT9S9aegmp85S/bReBVTb4QTFaRNptJ9kuYaNhnbNEOkbKb/A=="
|
||||
},
|
||||
"verror": {
|
||||
"version": "1.10.0",
|
||||
"resolved": "https://registry.npmjs.org/verror/-/verror-1.10.0.tgz",
|
||||
"integrity": "sha1-OhBcoXBTr1XW4nDB+CiGguGNpAA=",
|
||||
"requires": {
|
||||
"assert-plus": "^1.0.0",
|
||||
"core-util-is": "1.0.2",
|
||||
"extsprintf": "^1.2.0"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,92 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"io"
|
||||
"net/http"
|
||||
"net/url"
|
||||
|
||||
"golang.org/x/net/html"
|
||||
)
|
||||
|
||||
// scrapeImageURLs fetches a page and returns the absolute URLs of every image
|
||||
// it references.
|
||||
func scrapeImageURLs(client *http.Client, pageURL *url.URL) ([]string, error) {
|
||||
resp, err := client.Get(pageURL.String())
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("fetching %s: %w", pageURL, err)
|
||||
}
|
||||
defer resp.Body.Close()
|
||||
|
||||
if resp.StatusCode != http.StatusOK {
|
||||
return nil, fmt.Errorf("fetching %s: unexpected status %s", pageURL, resp.Status)
|
||||
}
|
||||
|
||||
return extractImageURLs(pageURL, resp.Body)
|
||||
}
|
||||
|
||||
// extractImageURLs pulls the src of every <img> in the document, resolves it
|
||||
// against base, and strips query strings so that cache-busting parameters do
|
||||
// not end up in file names. Duplicates and non-absolute results are dropped.
|
||||
func extractImageURLs(base *url.URL, r io.Reader) ([]string, error) {
|
||||
doc, err := html.Parse(r)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("parsing HTML: %w", err)
|
||||
}
|
||||
|
||||
var (
|
||||
urls []string
|
||||
seen = make(map[string]bool)
|
||||
walk func(*html.Node)
|
||||
)
|
||||
|
||||
walk = func(n *html.Node) {
|
||||
if n.Type == html.ElementNode && n.Data == "img" {
|
||||
if src, ok := attr(n, "src"); ok {
|
||||
if resolved, ok := resolveImageURL(base, src); ok && !seen[resolved] {
|
||||
seen[resolved] = true
|
||||
urls = append(urls, resolved)
|
||||
}
|
||||
}
|
||||
}
|
||||
for child := n.FirstChild; child != nil; child = child.NextSibling {
|
||||
walk(child)
|
||||
}
|
||||
}
|
||||
walk(doc)
|
||||
|
||||
return urls, nil
|
||||
}
|
||||
|
||||
// attr returns the value of the named attribute, if the node has it.
|
||||
func attr(n *html.Node, name string) (string, bool) {
|
||||
for _, a := range n.Attr {
|
||||
if a.Key == name {
|
||||
return a.Val, true
|
||||
}
|
||||
}
|
||||
return "", false
|
||||
}
|
||||
|
||||
// resolveImageURL makes src absolute relative to base and reports whether the
|
||||
// result is a usable http(s) URL.
|
||||
func resolveImageURL(base *url.URL, src string) (string, bool) {
|
||||
ref, err := url.Parse(src)
|
||||
if err != nil {
|
||||
return "", false
|
||||
}
|
||||
|
||||
resolved := base.ResolveReference(ref)
|
||||
// Drop the query so URLs like '/hsts-pixel.gif?c=3.2.5' yield clean names.
|
||||
resolved.RawQuery = ""
|
||||
resolved.Fragment = ""
|
||||
|
||||
if resolved.Host == "" {
|
||||
return "", false
|
||||
}
|
||||
// data: and other schemes are not downloadable files.
|
||||
if resolved.Scheme != "http" && resolved.Scheme != "https" {
|
||||
return "", false
|
||||
}
|
||||
return resolved.String(), true
|
||||
}
|
||||
Reference in new issue
Block a user