From fb0964bb6e84b4a18d47e7eb367eb29b24b5dbc5 Mon Sep 17 00:00:00 2001 From: tiennm99 Date: Sat, 25 Jul 2026 19:41:43 +0700 Subject: [PATCH] feat: replace Python and JS downloaders with a Go crawler Port both scripts into one Go program with a command per behavior: gallery walks the numbered ghibli.jp film galleries, scrape downloads every image referenced by a page. Downloads now run through a bounded worker pool, and non-200 responses are no longer written to disk as image files. Drop the orphaned package-lock.json, which pinned the deprecated request dependency and was the source of the repository's Dependabot alerts. --- ghibli-gallery-crawler/.gitignore | 11 + ghibli-gallery-crawler/README.md | 61 +++- ghibli-gallery-crawler/client.go | 36 +++ ghibli-gallery-crawler/crawler_test.go | 112 +++++++ ghibli-gallery-crawler/download.go | 157 ++++++++++ ghibli-gallery-crawler/download_images.js | 54 ---- ghibli-gallery-crawler/download_images.py | 90 ------ ghibli-gallery-crawler/gallery.go | 34 +++ ghibli-gallery-crawler/go.mod | 5 + ghibli-gallery-crawler/go.sum | 2 + ghibli-gallery-crawler/main.go | 204 +++++++++++++ ghibli-gallery-crawler/package-lock.json | 337 ---------------------- ghibli-gallery-crawler/scrape.go | 92 ++++++ 13 files changed, 703 insertions(+), 492 deletions(-) create mode 100644 ghibli-gallery-crawler/.gitignore create mode 100644 ghibli-gallery-crawler/client.go create mode 100644 ghibli-gallery-crawler/crawler_test.go create mode 100644 ghibli-gallery-crawler/download.go delete mode 100644 ghibli-gallery-crawler/download_images.js delete mode 100644 ghibli-gallery-crawler/download_images.py create mode 100644 ghibli-gallery-crawler/gallery.go create mode 100644 ghibli-gallery-crawler/go.mod create mode 100644 ghibli-gallery-crawler/go.sum create mode 100644 ghibli-gallery-crawler/main.go delete mode 100644 ghibli-gallery-crawler/package-lock.json create mode 100644 ghibli-gallery-crawler/scrape.go diff --git a/ghibli-gallery-crawler/.gitignore b/ghibli-gallery-crawler/.gitignore new file mode 100644 index 0000000..d95f7dc --- /dev/null +++ b/ghibli-gallery-crawler/.gitignore @@ -0,0 +1,11 @@ +# Build output +/ghibli-gallery-crawler +/ghibli-gallery-crawler.exe + +# Downloaded images +*.jpg +*.jpeg +*.png +*.gif +*.webp +*.svg diff --git a/ghibli-gallery-crawler/README.md b/ghibli-gallery-crawler/README.md index ec25b54..dd58ee0 100644 --- a/ghibli-gallery-crawler/README.md +++ b/ghibli-gallery-crawler/README.md @@ -1,27 +1,66 @@ # ghibli-gallery-crawler -Crawler for the Ghibli film gallery images (from 2020-08-31). Both Python and JavaScript variants included. +Image downloader in Go (originally written in Python and JavaScript in 2020-08-31). + +## Install + +```bash +go build -o ghibli-gallery-crawler . +``` + +Or run without building: `go run . `. ## Usage -```bash -# Python -python3 download_images.py +Two commands, one per behavior carried over from the original scripts. -# JavaScript -npm install -node download_images.js +### `gallery` — numbered ghibli.jp galleries + +Downloads `001.jpg` … `050.jpg` for each film, into one directory per film: + +```bash +# All 8 default films into ./marnie, ./kaguyahime, ... +ghibli-gallery-crawler gallery + +# Pick films, image count, and output location +ghibli-gallery-crawler gallery -films ponyo,chihiro -count 50 -out ./images ``` -Edit the script to set the source URLs and target directory before running. +| Flag | Default | Meaning | +|---|---|---| +| `-films` | the 8 films below | comma-separated film slugs | +| `-count` | `50` | highest image number tried per film | +| `-out` | `.` | where the per-film directories are created | +| `-base` | `https://www.ghibli.jp/gallery/` | base gallery URL | +| `-concurrency` | `8` | parallel downloads | +| `-timeout` | `30s` | per-request timeout | +| `-user-agent` | tool identifier | `User-Agent` header to send | -## Input format +Default films: `marnie`, `kaguyahime`, `kazetachinu`, `kokurikozaka`, `karigurashi`, `ponyo`, `ged`, `chihiro`. -Both variants read a hardcoded list of image URLs defined at the top of each script. Edit the URL array in `download_images.py` or `download_images.js` to point at your images and set the output directory path. +Image numbers are a fixed range, so films with fewer than `-count` images are normal — those are reported as `missing`, not as failures. + +### `scrape` — every image on a page + +Parses a page's HTML, collects every ``, resolves relative URLs, and downloads them: + +```bash +# Saves into ./en.wikipedia.org (the URL's host) +ghibli-gallery-crawler scrape https://en.wikipedia.org/wiki/Studio_Ghibli + +# Explicit output directory +ghibli-gallery-crawler scrape https://example.com -path ./images +``` + +Flags: `-path` (default: the URL's host), plus `-concurrency`, `-timeout`, and `-user-agent` as above. + +Query strings are stripped from image URLs so that names like `/hsts-pixel.gif?c=3.2.5` produce clean file names. `data:` URIs and duplicates are skipped. ## Output behavior -Images are saved into the configured output directory, named by their original filename from the URL. Existing files are overwritten. The Python variant uses `requests`; the JS variant uses `axios` + `fs.createWriteStream`. +Images are named after the last path segment of their URL. Existing files with the same name are overwritten. Missing directories are created. Non-200 responses are never written to disk, so error pages cannot land in your output as `.jpg` files. + +Note that `ghibli.jp` serves gallery images to any client but returns `403` for its HTML pages unless the request looks like a browser, so `gallery` works there while `scrape` does not. ## License diff --git a/ghibli-gallery-crawler/client.go b/ghibli-gallery-crawler/client.go new file mode 100644 index 0000000..d5065df --- /dev/null +++ b/ghibli-gallery-crawler/client.go @@ -0,0 +1,36 @@ +package main + +import ( + "net/http" + "time" +) + +// defaultUserAgent identifies this tool to servers. Some sites reject requests +// that do not send a User-Agent header at all, and an honest identifier lets +// operators see who is fetching. Override it with -user-agent. +const defaultUserAgent = "ghibli-gallery-crawler/1.0 (+https://github.com/tiennm99/ghibli-gallery-crawler)" + +// userAgentTransport stamps every request with a User-Agent header. +type userAgentTransport struct { + userAgent string + base http.RoundTripper +} + +func (t *userAgentTransport) RoundTrip(req *http.Request) (*http.Response, error) { + // Clone before mutating: RoundTrippers must not modify the caller's request. + clone := req.Clone(req.Context()) + clone.Header.Set("User-Agent", t.userAgent) + return t.base.RoundTrip(clone) +} + +// newClient builds an HTTP client that applies the given timeout to each +// request and identifies itself with userAgent. +func newClient(timeout time.Duration, userAgent string) *http.Client { + return &http.Client{ + Timeout: timeout, + Transport: &userAgentTransport{ + userAgent: userAgent, + base: http.DefaultTransport, + }, + } +} diff --git a/ghibli-gallery-crawler/crawler_test.go b/ghibli-gallery-crawler/crawler_test.go new file mode 100644 index 0000000..285eff1 --- /dev/null +++ b/ghibli-gallery-crawler/crawler_test.go @@ -0,0 +1,112 @@ +package main + +import ( + "net/url" + "path/filepath" + "strings" + "testing" +) + +func TestExtractImageURLs(t *testing.T) { + base, err := url.Parse("https://example.com/gallery/index.html") + if err != nil { + t.Fatalf("parsing base: %v", err) + } + + doc := ` + + + + + + no src attribute + + ` + + got, err := extractImageURLs(base, strings.NewReader(doc)) + if err != nil { + t.Fatalf("extractImageURLs: %v", err) + } + + want := []string{ + "https://example.com/gallery/photo.jpg", + "https://example.com/absolute/other.png", + "https://cdn.example.net/remote.gif", + "https://example.com/hsts-pixel.gif", // query stripped + } + + if len(got) != len(want) { + t.Fatalf("got %d URLs %v, want %d %v", len(got), got, len(want), want) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("URL %d = %q, want %q", i, got[i], want[i]) + } + } +} + +func TestGalleryJobs(t *testing.T) { + jobs, err := galleryJobs("https://www.ghibli.jp/gallery/", []string{"ponyo", "ged"}, 3, "out") + if err != nil { + t.Fatalf("galleryJobs: %v", err) + } + + if len(jobs) != 6 { + t.Fatalf("got %d jobs, want 6", len(jobs)) + } + + // Numbers are zero-padded to three digits, as in the original script. + if want := "https://www.ghibli.jp/gallery/ponyo001.jpg"; jobs[0].url != want { + t.Errorf("first URL = %q, want %q", jobs[0].url, want) + } + if want := filepath.Join("out", "ponyo"); jobs[0].dir != want { + t.Errorf("first dir = %q, want %q", jobs[0].dir, want) + } + if want := "https://www.ghibli.jp/gallery/ged003.jpg"; jobs[5].url != want { + t.Errorf("last URL = %q, want %q", jobs[5].url, want) + } +} + +func TestFileNameFromURL(t *testing.T) { + cases := []struct { + in string + want string + wantErr bool + }{ + {in: "https://example.com/a/b/chihiro001.jpg", want: "chihiro001.jpg"}, + {in: "https://example.com/image.png", want: "image.png"}, + {in: "https://example.com/", wantErr: true}, + {in: "https://example.com", wantErr: true}, + } + + for _, c := range cases { + got, err := fileNameFromURL(c.in) + if c.wantErr { + if err == nil { + t.Errorf("fileNameFromURL(%q) = %q, want error", c.in, got) + } + continue + } + if err != nil { + t.Errorf("fileNameFromURL(%q): %v", c.in, err) + continue + } + if got != c.want { + t.Errorf("fileNameFromURL(%q) = %q, want %q", c.in, got, c.want) + } + } +} + +func TestSplitFilms(t *testing.T) { + got := splitFilms(" ponyo , ,ged, ../escape ") + want := []string{"ponyo", "ged"} + + if len(got) != len(want) { + t.Fatalf("got %v, want %v", got, want) + } + for i := range want { + if got[i] != want[i] { + t.Errorf("slug %d = %q, want %q", i, got[i], want[i]) + } + } +} diff --git a/ghibli-gallery-crawler/download.go b/ghibli-gallery-crawler/download.go new file mode 100644 index 0000000..d97f138 --- /dev/null +++ b/ghibli-gallery-crawler/download.go @@ -0,0 +1,157 @@ +package main + +import ( + "errors" + "fmt" + "io" + "net/http" + "net/url" + "os" + "path" + "path/filepath" + "sync" +) + +// job is a single image to fetch and the directory to store it in. +type job struct { + url string + dir string +} + +// errNotFound marks images the server does not have. The numbered gallery mode +// always asks for a fixed range, so missing images are expected, not failures. +var errNotFound = errors.New("not found") + +// download fetches every job using at most concurrency parallel requests and +// reports how many succeeded. Individual failures are logged and do not stop +// the run. +func download(client *http.Client, jobs []job, concurrency int) error { + if concurrency < 1 { + concurrency = 1 + } + if concurrency > len(jobs) { + concurrency = len(jobs) + } + + // Create directories up front so parallel workers never race on MkdirAll. + for _, j := range jobs { + if err := os.MkdirAll(j.dir, 0o755); err != nil { + return fmt.Errorf("creating %s: %w", j.dir, err) + } + } + + var ( + queue = make(chan job) + wg sync.WaitGroup + mu sync.Mutex + saved int + missing int + failed int + ) + + for range concurrency { + wg.Add(1) + go func() { + defer wg.Done() + for j := range queue { + size, err := downloadOne(client, j) + + mu.Lock() + switch { + case err == nil: + saved++ + fmt.Printf("saved %s (%s)\n", j.url, humanSize(size)) + case errors.Is(err, errNotFound): + missing++ + default: + failed++ + fmt.Fprintf(os.Stderr, "failed %s: %v\n", j.url, err) + } + mu.Unlock() + } + }() + } + + for _, j := range jobs { + queue <- j + } + close(queue) + wg.Wait() + + fmt.Printf("Done: %d saved, %d missing, %d failed\n", saved, missing, failed) + if saved == 0 && failed > 0 { + return fmt.Errorf("every download failed") + } + return nil +} + +// downloadOne writes a single image to disk and returns the bytes written. An +// existing file with the same name is overwritten. +func downloadOne(client *http.Client, j job) (int64, error) { + name, err := fileNameFromURL(j.url) + if err != nil { + return 0, err + } + + resp, err := client.Get(j.url) + if err != nil { + return 0, err + } + defer resp.Body.Close() + + if resp.StatusCode == http.StatusNotFound || resp.StatusCode == http.StatusForbidden { + return 0, errNotFound + } + if resp.StatusCode != http.StatusOK { + // Bail out before writing, so error pages never land on disk as images. + return 0, fmt.Errorf("unexpected status %s", resp.Status) + } + + dest := filepath.Join(j.dir, name) + f, err := os.Create(dest) + if err != nil { + return 0, err + } + defer f.Close() + + written, err := io.Copy(f, resp.Body) + if err != nil { + // Leave no truncated file behind on a mid-transfer error. + os.Remove(dest) + return 0, err + } + return written, nil +} + +// fileNameFromURL derives a safe local file name from the last path segment of +// an image URL. +func fileNameFromURL(rawURL string) (string, error) { + parsed, err := url.Parse(rawURL) + if err != nil { + return "", err + } + + name := path.Base(parsed.Path) + // path.Base returns "." or "/" when there is nothing usable to take. + if name == "" || name == "." || name == "/" || name == ".." { + return "", fmt.Errorf("cannot derive a file name from %q", rawURL) + } + // Guard against a remote path smuggling in separators. + return filepath.Base(name), nil +} + +// humanSize formats a byte count for display. +func humanSize(n int64) string { + const unit = 1024 + if n < unit { + return fmt.Sprintf("%d B", n) + } + value := float64(n) + for _, suffix := range []string{"KiB", "MiB", "GiB"} { + value /= unit + if value < unit { + return fmt.Sprintf("%.1f %s", value, suffix) + } + } + return fmt.Sprintf("%.1f TiB", value) +} diff --git a/ghibli-gallery-crawler/download_images.js b/ghibli-gallery-crawler/download_images.js deleted file mode 100644 index cffab3f..0000000 --- a/ghibli-gallery-crawler/download_images.js +++ /dev/null @@ -1,54 +0,0 @@ -const fs = require('fs') -const request = require('request') -const { - cookieCompare -} = require('tough-cookie') - -const download = (url, path, callback) => { - request.head(url, (err, res, body) => { - request(url) - .pipe(fs.createWriteStream(path)) - .on('close', callback) - }) -} - -// const url = 'https://…' -// const path = './images/image.png' - -// download(url, path, () => { -// console.log('✅ Done!') -// }) - -// let urlConst = "https://www.iamag.co/wp-content/uploads/2017/07/yourname-background" -// for (let i = 1; i < 200; i++) { -// download(urlConst + i + ".jpeg", "img/" + i + ".jpeg", () => { -// console.log("Done: " + i); -// }) -// } - -function lpad(n, width, z) { - z = z || '0'; - n = n + ''; - return n.length >= width ? n : new Array(width - n.length + 1).join(z) + n; -} - -let ghibli = "http://www.ghibli.jp/gallery/"; -let listFilm = ["marnie", "kaguyahime", "kazetachinu", "kokurikozaka", "karigurashi", "ponyo", "ged", "chihiro"]; -listFilm.forEach(film => { - let filmDir = "./" + film; - fs.mkdir(filmDir, function(err) { - if (err) { - console.log(err); - } else { - console.log("Created " + filmDir); - } - }); - for (let i = 1; i <= 50; i++) { - let imgName = lpad(i, 3) + ".jpg"; - let imgUrl = ghibli + film + imgName; - console.log(imgUrl); - download(imgUrl, filmDir + "/" + imgName, () => { - console.log("Done: " + imgName); - }) - } -}) \ No newline at end of file diff --git a/ghibli-gallery-crawler/download_images.py b/ghibli-gallery-crawler/download_images.py deleted file mode 100644 index 084069b..0000000 --- a/ghibli-gallery-crawler/download_images.py +++ /dev/null @@ -1,90 +0,0 @@ -import requests -import os -from tqdm import tqdm -from bs4 import BeautifulSoup as bs -from urllib.parse import urljoin, urlparse - - -def is_valid(url): - """ - Checks whether `url` is a valid URL. - """ - parsed = urlparse(url) - return bool(parsed.netloc) and bool(parsed.scheme) - - -def get_all_images(url): - """ - Returns all image URLs on a single `url` - """ - soup = bs(requests.get(url).content, "html.parser") - urls = [] - for img in tqdm(soup.find_all("img"), "Extracting images"): - img_url = img.attrs.get("src") - if not img_url: - # if img does not contain src attribute, just skip - continue - # make the URL absolute by joining domain with the URL that is just extracted - img_url = urljoin(url, img_url) - # remove URLs like '/hsts-pixel.gif?c=3.2.5' - try: - pos = img_url.index("?") - img_url = img_url[:pos] - except ValueError: - pass - # finally, if the url is valid - if is_valid(img_url): - urls.append(img_url) - return urls - - -def download(url, pathname): - """ - Downloads a file given an URL and puts it in the folder `pathname` - """ - # if path doesn't exist, make that path dir - if not os.path.isdir(pathname): - os.makedirs(pathname) - # download the body of response by chunk, not immediately - response = requests.get(url, stream=True) - - # get the total file size - file_size = int(response.headers.get("Content-Length", 0)) - - # get the file name - filename = os.path.join(pathname, url.split("/")[-1]) - - # progress bar, changing the unit to bytes instead of iteration (default by tqdm) - progress = tqdm(response.iter_content(1024), f"Downloading {filename}", total=file_size, unit="B", unit_scale=True, unit_divisor=1024) - with open(filename, "wb") as f: - for data in progress: - # write data read to the file - f.write(data) - # update the progress bar manually - progress.update(len(data)) - - -def main(url, path): - # get all images - imgs = get_all_images(url) - for img in imgs: - # for each img, download it - download(img, path) - - - -if __name__ == "__main__": - import argparse - parser = argparse.ArgumentParser(description="This script downloads all images from a web page") - parser.add_argument("url", help="The URL of the web page you want to download images") - parser.add_argument("-p", "--path", help="The Directory you want to store your images, default is the domain of URL passed") - - args = parser.parse_args() - url = args.url - path = args.path - - if not path: - # if path isn't specified, use the domain name of that url as the folder name - path = urlparse(url).netloc - - main(url, path) \ No newline at end of file diff --git a/ghibli-gallery-crawler/gallery.go b/ghibli-gallery-crawler/gallery.go new file mode 100644 index 0000000..72ba2c4 --- /dev/null +++ b/ghibli-gallery-crawler/gallery.go @@ -0,0 +1,34 @@ +package main + +import ( + "fmt" + "net/url" + "path/filepath" +) + +// galleryJobs builds the download list for the numbered ghibli.jp galleries: +// for each film, images are named 001.jpg through .jpg and +// are stored in a directory named after the film. +func galleryJobs(base string, films []string, count int, out string) ([]job, error) { + baseURL, err := url.Parse(base) + if err != nil { + return nil, fmt.Errorf("parsing base URL %q: %w", base, err) + } + if baseURL.Scheme == "" || baseURL.Host == "" { + return nil, fmt.Errorf("base URL %q must be absolute", base) + } + + jobs := make([]job, 0, len(films)*count) + for _, film := range films { + dir := filepath.Join(out, film) + for i := 1; i <= count; i++ { + name := fmt.Sprintf("%s%03d.jpg", film, i) + imageURL, err := url.JoinPath(base, name) + if err != nil { + return nil, fmt.Errorf("building URL for %s: %w", name, err) + } + jobs = append(jobs, job{url: imageURL, dir: dir}) + } + } + return jobs, nil +} diff --git a/ghibli-gallery-crawler/go.mod b/ghibli-gallery-crawler/go.mod new file mode 100644 index 0000000..a0e2a5c --- /dev/null +++ b/ghibli-gallery-crawler/go.mod @@ -0,0 +1,5 @@ +module github.com/tiennm99/ghibli-gallery-crawler + +go 1.26.4 + +require golang.org/x/net v0.57.0 diff --git a/ghibli-gallery-crawler/go.sum b/ghibli-gallery-crawler/go.sum new file mode 100644 index 0000000..d902440 --- /dev/null +++ b/ghibli-gallery-crawler/go.sum @@ -0,0 +1,2 @@ +golang.org/x/net v0.57.0 h1:K5+3DljvIuDG9/Jv9rvyMywYNFCQ9RSUY6OOTTkT+tE= +golang.org/x/net v0.57.0/go.mod h1:KpXc8iv+r3XplLAG/f7Jsf9RPszJzdR0f58q9vGOuEU= diff --git a/ghibli-gallery-crawler/main.go b/ghibli-gallery-crawler/main.go new file mode 100644 index 0000000..30bbefa --- /dev/null +++ b/ghibli-gallery-crawler/main.go @@ -0,0 +1,204 @@ +// Command ghibli-gallery-crawler downloads images from web pages. +// +// It offers two modes, mirroring the two scripts it replaces: +// +// scrape - parse a page's tags and download every image found +// gallery - walk the numbered per-film galleries on ghibli.jp +package main + +import ( + "flag" + "fmt" + "net/url" + "os" + "path/filepath" + "strings" + "time" +) + +const ( + // defaultGalleryBase is where the numbered film galleries live. The + // original script used http://; https:// reaches the same files. + defaultGalleryBase = "https://www.ghibli.jp/gallery/" + // defaultGalleryCount is the highest image number tried per film. + defaultGalleryCount = 50 +) + +// defaultFilms lists the gallery slugs the original JavaScript script crawled. +var defaultFilms = []string{ + "marnie", "kaguyahime", "kazetachinu", "kokurikozaka", + "karigurashi", "ponyo", "ged", "chihiro", +} + +func main() { + if err := run(os.Args[1:]); err != nil { + fmt.Fprintln(os.Stderr, "error:", err) + os.Exit(1) + } +} + +func run(args []string) error { + if len(args) == 0 { + usage() + return fmt.Errorf("no command given") + } + + switch args[0] { + case "scrape": + return runScrape(args[1:]) + case "gallery": + return runGallery(args[1:]) + case "help", "-h", "--help": + usage() + return nil + default: + usage() + return fmt.Errorf("unknown command %q", args[0]) + } +} + +func usage() { + fmt.Fprint(os.Stderr, `ghibli-gallery-crawler downloads images from web pages. + +Usage: + ghibli-gallery-crawler scrape [flags] + ghibli-gallery-crawler gallery [flags] + +Commands: + scrape Download every found on a single page. + gallery Download the numbered ghibli.jp galleries for a list of films. + +Run a command with -h to see its flags. +`) +} + +// runScrape downloads every image referenced by a single page. The default +// output directory is the page's host, matching the Python script it replaces. +func runScrape(args []string) error { + fs := flag.NewFlagSet("scrape", flag.ExitOnError) + path := fs.String("path", "", "directory to store images in (default: the URL's host)") + concurrency := fs.Int("concurrency", 8, "number of parallel downloads") + timeout := fs.Duration("timeout", 30*time.Second, "per-request timeout") + userAgent := fs.String("user-agent", defaultUserAgent, "User-Agent header to send") + fs.Usage = func() { + fmt.Fprintln(fs.Output(), "Usage: ghibli-gallery-crawler scrape [flags]") + fs.PrintDefaults() + } + positional, err := parseWithPositionals(fs, args) + if err != nil { + return err + } + + if len(positional) != 1 { + fs.Usage() + return fmt.Errorf("scrape needs exactly one URL argument, got %d", len(positional)) + } + pageURL := positional[0] + + parsed, err := url.Parse(pageURL) + if err != nil { + return fmt.Errorf("parsing %q: %w", pageURL, err) + } + if parsed.Scheme == "" || parsed.Host == "" { + return fmt.Errorf("%q is not an absolute URL", pageURL) + } + + dir := *path + if dir == "" { + dir = parsed.Host + } + + client := newClient(*timeout, *userAgent) + + imageURLs, err := scrapeImageURLs(client, parsed) + if err != nil { + return err + } + if len(imageURLs) == 0 { + fmt.Printf("No images found on %s\n", pageURL) + return nil + } + fmt.Printf("Found %d image(s) on %s\n", len(imageURLs), pageURL) + + jobs := make([]job, 0, len(imageURLs)) + for _, imageURL := range imageURLs { + jobs = append(jobs, job{url: imageURL, dir: dir}) + } + return download(client, jobs, *concurrency) +} + +// runGallery downloads images named 001.jpg .. NNN.jpg for each +// film, storing each film's images in its own directory. This mirrors the +// JavaScript script it replaces. +func runGallery(args []string) error { + fs := flag.NewFlagSet("gallery", flag.ExitOnError) + base := fs.String("base", defaultGalleryBase, "base gallery URL") + films := fs.String("films", strings.Join(defaultFilms, ","), "comma-separated film slugs") + count := fs.Int("count", defaultGalleryCount, "highest image number to try per film") + out := fs.String("out", ".", "directory to create the per-film directories in") + concurrency := fs.Int("concurrency", 8, "number of parallel downloads") + timeout := fs.Duration("timeout", 30*time.Second, "per-request timeout") + userAgent := fs.String("user-agent", defaultUserAgent, "User-Agent header to send") + fs.Usage = func() { + fmt.Fprintln(fs.Output(), "Usage: ghibli-gallery-crawler gallery [flags]") + fs.PrintDefaults() + } + if err := fs.Parse(args); err != nil { + return err + } + + if *count < 1 { + return fmt.Errorf("-count must be at least 1, got %d", *count) + } + + slugs := splitFilms(*films) + if len(slugs) == 0 { + return fmt.Errorf("-films must name at least one film") + } + + jobs, err := galleryJobs(*base, slugs, *count, *out) + if err != nil { + return err + } + fmt.Printf("Trying %d image(s) across %d film(s)\n", len(jobs), len(slugs)) + + client := newClient(*timeout, *userAgent) + return download(client, jobs, *concurrency) +} + +// parseWithPositionals parses flags that may appear before or after positional +// arguments. The standard flag package stops at the first non-flag argument, so +// the remainder is fed back through the parser one positional at a time. +func parseWithPositionals(fs *flag.FlagSet, args []string) ([]string, error) { + var positional []string + for len(args) > 0 { + if err := fs.Parse(args); err != nil { + return nil, err + } + args = fs.Args() + if len(args) > 0 { + positional = append(positional, args[0]) + args = args[1:] + } + } + return positional, nil +} + +// splitFilms turns a comma-separated flag value into clean slugs, dropping +// empty entries and anything that would escape the output directory. +func splitFilms(value string) []string { + var slugs []string + for _, raw := range strings.Split(value, ",") { + slug := strings.TrimSpace(raw) + if slug == "" { + continue + } + // A slug becomes a directory name, so reject path separators. + if slug != filepath.Base(slug) || slug == "." || slug == ".." { + fmt.Fprintf(os.Stderr, "skipping invalid film slug %q\n", slug) + continue + } + slugs = append(slugs, slug) + } + return slugs +} diff --git a/ghibli-gallery-crawler/package-lock.json b/ghibli-gallery-crawler/package-lock.json deleted file mode 100644 index b8e2660..0000000 --- a/ghibli-gallery-crawler/package-lock.json +++ /dev/null @@ -1,337 +0,0 @@ -{ - "requires": true, - "lockfileVersion": 1, - "dependencies": { - "ajv": { - "version": "6.12.4", - "resolved": "https://registry.npmjs.org/ajv/-/ajv-6.12.4.tgz", - "integrity": "sha512-eienB2c9qVQs2KWexhkrdMLVDoIQCz5KSeLxwg9Lzk4DOfBtIK9PQwwufcsn1jjGuf9WZmqPMbGxOzfcuphJCQ==", - "requires": { - "fast-deep-equal": "^3.1.1", - "fast-json-stable-stringify": "^2.0.0", - "json-schema-traverse": "^0.4.1", - "uri-js": "^4.2.2" - } - }, - "asn1": { - "version": "0.2.4", - "resolved": "https://registry.npmjs.org/asn1/-/asn1-0.2.4.tgz", - "integrity": "sha512-jxwzQpLQjSmWXgwaCZE9Nz+glAG01yF1QnWgbhGwHI5A6FRIEY6IVqtHhIepHqI7/kyEyQEagBC5mBEFlIYvdg==", - "requires": { - "safer-buffer": "~2.1.0" - } - }, - "assert-plus": { - "version": "1.0.0", - "resolved": "https://registry.npmjs.org/assert-plus/-/assert-plus-1.0.0.tgz", - "integrity": "sha1-8S4PPF13sLHN2RRpQuTpbB5N1SU=" - }, - "asynckit": { - "version": "0.4.0", - "resolved": "https://registry.npmjs.org/asynckit/-/asynckit-0.4.0.tgz", - "integrity": "sha1-x57Zf380y48robyXkLzDZkdLS3k=" - }, - "aws-sign2": { - "version": "0.7.0", - "resolved": "https://registry.npmjs.org/aws-sign2/-/aws-sign2-0.7.0.tgz", - "integrity": "sha1-tG6JCTSpWR8tL2+G1+ap8bP+dqg=" - }, - "aws4": { - "version": "1.10.1", - "resolved": "https://registry.npmjs.org/aws4/-/aws4-1.10.1.tgz", - "integrity": "sha512-zg7Hz2k5lI8kb7U32998pRRFin7zJlkfezGJjUc2heaD4Pw2wObakCDVzkKztTm/Ln7eiVvYsjqak0Ed4LkMDA==" - }, - "bcrypt-pbkdf": { - "version": "1.0.2", - "resolved": "https://registry.npmjs.org/bcrypt-pbkdf/-/bcrypt-pbkdf-1.0.2.tgz", - "integrity": "sha1-pDAdOJtqQ/m2f/PKEaP2Y342Dp4=", - "requires": { - "tweetnacl": "^0.14.3" - } - }, - "caseless": { - "version": "0.12.0", - "resolved": "https://registry.npmjs.org/caseless/-/caseless-0.12.0.tgz", - "integrity": "sha1-G2gcIf+EAzyCZUMJBolCDRhxUdw=" - }, - "combined-stream": { - "version": "1.0.8", - "resolved": "https://registry.npmjs.org/combined-stream/-/combined-stream-1.0.8.tgz", - "integrity": "sha512-FQN4MRfuJeHf7cBbBMJFXhKSDq+2kAArBlmRBvcvFE5BB1HZKXtSFASDhdlz9zOYwxh8lDdnvmMOe/+5cdoEdg==", - "requires": { - "delayed-stream": "~1.0.0" - } - }, - "core-util-is": { - "version": "1.0.2", - "resolved": "https://registry.npmjs.org/core-util-is/-/core-util-is-1.0.2.tgz", - "integrity": "sha1-tf1UIgqivFq1eqtxQMlAdUUDwac=" - }, - "dashdash": { - "version": "1.14.1", - "resolved": "https://registry.npmjs.org/dashdash/-/dashdash-1.14.1.tgz", - "integrity": "sha1-hTz6D3y+L+1d4gMmuN1YEDX24vA=", - "requires": { - "assert-plus": "^1.0.0" - } - }, - "delayed-stream": { - "version": "1.0.0", - "resolved": "https://registry.npmjs.org/delayed-stream/-/delayed-stream-1.0.0.tgz", - "integrity": "sha1-3zrhmayt+31ECqrgsp4icrJOxhk=" - }, - "ecc-jsbn": { - "version": "0.1.2", - "resolved": "https://registry.npmjs.org/ecc-jsbn/-/ecc-jsbn-0.1.2.tgz", - "integrity": "sha1-OoOpBOVDUyh4dMVkt1SThoSamMk=", - "requires": { - "jsbn": "~0.1.0", - "safer-buffer": "^2.1.0" - } - }, - "extend": { - "version": "3.0.2", - "resolved": "https://registry.npmjs.org/extend/-/extend-3.0.2.tgz", - "integrity": "sha512-fjquC59cD7CyW6urNXK0FBufkZcoiGG80wTuPujX590cB5Ttln20E2UB4S/WARVqhXffZl2LNgS+gQdPIIim/g==" - }, - "extsprintf": { - "version": "1.3.0", - "resolved": "https://registry.npmjs.org/extsprintf/-/extsprintf-1.3.0.tgz", - "integrity": "sha1-lpGEQOMEGnpBT4xS48V06zw+HgU=" - }, - "fast-deep-equal": { - "version": "3.1.3", - "resolved": "https://registry.npmjs.org/fast-deep-equal/-/fast-deep-equal-3.1.3.tgz", - "integrity": "sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q==" - }, - "fast-json-stable-stringify": { - "version": "2.1.0", - "resolved": "https://registry.npmjs.org/fast-json-stable-stringify/-/fast-json-stable-stringify-2.1.0.tgz", - "integrity": "sha512-lhd/wF+Lk98HZoTCtlVraHtfh5XYijIjalXck7saUtuanSDyLMxnHhSXEDJqHxD7msR8D0uCmqlkwjCV8xvwHw==" - }, - "forever-agent": { - "version": "0.6.1", - "resolved": "https://registry.npmjs.org/forever-agent/-/forever-agent-0.6.1.tgz", - "integrity": "sha1-+8cfDEGt6zf5bFd60e1C2P2sypE=" - }, - "form-data": { - "version": "2.3.3", - "resolved": "https://registry.npmjs.org/form-data/-/form-data-2.3.3.tgz", - "integrity": "sha512-1lLKB2Mu3aGP1Q/2eCOx0fNbRMe7XdwktwOruhfqqd0rIJWwN4Dh+E3hrPSlDCXnSR7UtZ1N38rVXm+6+MEhJQ==", - "requires": { - "asynckit": "^0.4.0", - "combined-stream": "^1.0.6", - "mime-types": "^2.1.12" - } - }, - "getpass": { - "version": "0.1.7", - "resolved": "https://registry.npmjs.org/getpass/-/getpass-0.1.7.tgz", - "integrity": "sha1-Xv+OPmhNVprkyysSgmBOi6YhSfo=", - "requires": { - "assert-plus": "^1.0.0" - } - }, - "har-schema": { - "version": "2.0.0", - "resolved": "https://registry.npmjs.org/har-schema/-/har-schema-2.0.0.tgz", - "integrity": "sha1-qUwiJOvKwEeCoNkDVSHyRzW37JI=" - }, - "har-validator": { - "version": "5.1.5", - "resolved": "https://registry.npmjs.org/har-validator/-/har-validator-5.1.5.tgz", - "integrity": "sha512-nmT2T0lljbxdQZfspsno9hgrG3Uir6Ks5afism62poxqBM6sDnMEuPmzTq8XN0OEwqKLLdh1jQI3qyE66Nzb3w==", - "requires": { - "ajv": "^6.12.3", - "har-schema": "^2.0.0" - } - }, - "http-signature": { - "version": "1.2.0", - "resolved": "https://registry.npmjs.org/http-signature/-/http-signature-1.2.0.tgz", - "integrity": "sha1-muzZJRFHcvPZW2WmCruPfBj7rOE=", - "requires": { - "assert-plus": "^1.0.0", - "jsprim": "^1.2.2", - "sshpk": "^1.7.0" - } - }, - "is-typedarray": { - "version": "1.0.0", - "resolved": "https://registry.npmjs.org/is-typedarray/-/is-typedarray-1.0.0.tgz", - "integrity": "sha1-5HnICFjfDBsR3dppQPlgEfzaSpo=" - }, - "isstream": { - "version": "0.1.2", - "resolved": "https://registry.npmjs.org/isstream/-/isstream-0.1.2.tgz", - "integrity": "sha1-R+Y/evVa+m+S4VAOaQ64uFKcCZo=" - }, - "jsbn": { - "version": "0.1.1", - "resolved": "https://registry.npmjs.org/jsbn/-/jsbn-0.1.1.tgz", - "integrity": "sha1-peZUwuWi3rXyAdls77yoDA7y9RM=" - }, - "json-schema": { - "version": "0.2.3", - "resolved": "https://registry.npmjs.org/json-schema/-/json-schema-0.2.3.tgz", - "integrity": "sha1-tIDIkuWaLwWVTOcnvT8qTogvnhM=" - }, - "json-schema-traverse": { - "version": "0.4.1", - "resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-0.4.1.tgz", - "integrity": "sha512-xbbCH5dCYU5T8LcEhhuh7HJ88HXuW3qsI3Y0zOZFKfZEHcpWiHU/Jxzk629Brsab/mMiHQti9wMP+845RPe3Vg==" - }, - "json-stringify-safe": { - "version": "5.0.1", - "resolved": "https://registry.npmjs.org/json-stringify-safe/-/json-stringify-safe-5.0.1.tgz", - "integrity": "sha1-Epai1Y/UXxmg9s4B1lcB4sc1tus=" - }, - "jsprim": { - "version": "1.4.1", - "resolved": "https://registry.npmjs.org/jsprim/-/jsprim-1.4.1.tgz", - "integrity": "sha1-MT5mvB5cwG5Di8G3SZwuXFastqI=", - "requires": { - "assert-plus": "1.0.0", - "extsprintf": "1.3.0", - "json-schema": "0.2.3", - "verror": "1.10.0" - } - }, - "mime-db": { - "version": "1.44.0", - "resolved": "https://registry.npmjs.org/mime-db/-/mime-db-1.44.0.tgz", - "integrity": "sha512-/NOTfLrsPBVeH7YtFPgsVWveuL+4SjjYxaQ1xtM1KMFj7HdxlBlxeyNLzhyJVx7r4rZGJAZ/6lkKCitSc/Nmpg==" - }, - "mime-types": { - "version": "2.1.27", - "resolved": "https://registry.npmjs.org/mime-types/-/mime-types-2.1.27.tgz", - "integrity": "sha512-JIhqnCasI9yD+SsmkquHBxTSEuZdQX5BuQnS2Vc7puQQQ+8yiP5AY5uWhpdv4YL4VM5c6iliiYWPgJ/nJQLp7w==", - "requires": { - "mime-db": "1.44.0" - } - }, - "oauth-sign": { - "version": "0.9.0", - "resolved": "https://registry.npmjs.org/oauth-sign/-/oauth-sign-0.9.0.tgz", - "integrity": "sha512-fexhUFFPTGV8ybAtSIGbV6gOkSv8UtRbDBnAyLQw4QPKkgNlsH2ByPGtMUqdWkos6YCRmAqViwgZrJc/mRDzZQ==" - }, - "performance-now": { - "version": "2.1.0", - "resolved": "https://registry.npmjs.org/performance-now/-/performance-now-2.1.0.tgz", - "integrity": "sha1-Ywn04OX6kT7BxpMHrjZLSzd8nns=" - }, - "psl": { - "version": "1.8.0", - "resolved": "https://registry.npmjs.org/psl/-/psl-1.8.0.tgz", - "integrity": "sha512-RIdOzyoavK+hA18OGGWDqUTsCLhtA7IcZ/6NCs4fFJaHBDab+pDDmDIByWFRQJq2Cd7r1OoQxBGKOaztq+hjIQ==" - }, - "punycode": { - "version": "2.1.1", - "resolved": "https://registry.npmjs.org/punycode/-/punycode-2.1.1.tgz", - "integrity": "sha512-XRsRjdf+j5ml+y/6GKHPZbrF/8p2Yga0JPtdqTIY2Xe5ohJPD9saDJJLPvp9+NSBprVvevdXZybnj2cv8OEd0A==" - }, - "qs": { - "version": "6.5.2", - "resolved": "https://registry.npmjs.org/qs/-/qs-6.5.2.tgz", - "integrity": "sha512-N5ZAX4/LxJmF+7wN74pUD6qAh9/wnvdQcjq9TZjevvXzSUo7bfmw91saqMjzGS2xq91/odN2dW/WOl7qQHNDGA==" - }, - "request": { - "version": "2.88.2", - "resolved": "https://registry.npmjs.org/request/-/request-2.88.2.tgz", - "integrity": "sha512-MsvtOrfG9ZcrOwAW+Qi+F6HbD0CWXEh9ou77uOb7FM2WPhwT7smM833PzanhJLsgXjN89Ir6V2PczXNnMpwKhw==", - "requires": { - "aws-sign2": "~0.7.0", - "aws4": "^1.8.0", - "caseless": "~0.12.0", - "combined-stream": "~1.0.6", - "extend": "~3.0.2", - "forever-agent": "~0.6.1", - "form-data": "~2.3.2", - "har-validator": "~5.1.3", - "http-signature": "~1.2.0", - "is-typedarray": "~1.0.0", - "isstream": "~0.1.2", - "json-stringify-safe": "~5.0.1", - "mime-types": "~2.1.19", - "oauth-sign": "~0.9.0", - "performance-now": "^2.1.0", - "qs": "~6.5.2", - "safe-buffer": "^5.1.2", - "tough-cookie": "~2.5.0", - "tunnel-agent": "^0.6.0", - "uuid": "^3.3.2" - } - }, - "safe-buffer": { - "version": "5.2.1", - "resolved": "https://registry.npmjs.org/safe-buffer/-/safe-buffer-5.2.1.tgz", - "integrity": "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ==" - }, - "safer-buffer": { - "version": "2.1.2", - "resolved": "https://registry.npmjs.org/safer-buffer/-/safer-buffer-2.1.2.tgz", - "integrity": "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg==" - }, - "sshpk": { - "version": "1.16.1", - "resolved": "https://registry.npmjs.org/sshpk/-/sshpk-1.16.1.tgz", - "integrity": "sha512-HXXqVUq7+pcKeLqqZj6mHFUMvXtOJt1uoUx09pFW6011inTMxqI8BA8PM95myrIyyKwdnzjdFjLiE6KBPVtJIg==", - "requires": { - "asn1": "~0.2.3", - "assert-plus": "^1.0.0", - "bcrypt-pbkdf": "^1.0.0", - "dashdash": "^1.12.0", - "ecc-jsbn": "~0.1.1", - "getpass": "^0.1.1", - "jsbn": "~0.1.0", - "safer-buffer": "^2.0.2", - "tweetnacl": "~0.14.0" - } - }, - "tough-cookie": { - "version": "2.5.0", - "resolved": "https://registry.npmjs.org/tough-cookie/-/tough-cookie-2.5.0.tgz", - "integrity": "sha512-nlLsUzgm1kfLXSXfRZMc1KLAugd4hqJHDTvc2hDIwS3mZAfMEuMbc03SujMF+GEcpaX/qboeycw6iO8JwVv2+g==", - "requires": { - "psl": "^1.1.28", - "punycode": "^2.1.1" - } - }, - "tunnel-agent": { - "version": "0.6.0", - "resolved": "https://registry.npmjs.org/tunnel-agent/-/tunnel-agent-0.6.0.tgz", - "integrity": "sha1-J6XeoGs2sEoKmWZ3SykIaPD8QP0=", - "requires": { - "safe-buffer": "^5.0.1" - } - }, - "tweetnacl": { - "version": "0.14.5", - "resolved": "https://registry.npmjs.org/tweetnacl/-/tweetnacl-0.14.5.tgz", - "integrity": "sha1-WuaBd/GS1EViadEIr6k/+HQ/T2Q=" - }, - "uri-js": { - "version": "4.4.0", - "resolved": "https://registry.npmjs.org/uri-js/-/uri-js-4.4.0.tgz", - "integrity": "sha512-B0yRTzYdUCCn9n+F4+Gh4yIDtMQcaJsmYBDsTSG8g/OejKBodLQ2IHfN3bM7jUsRXndopT7OIXWdYqc1fjmV6g==", - "requires": { - "punycode": "^2.1.0" - } - }, - "uuid": { - "version": "3.4.0", - "resolved": "https://registry.npmjs.org/uuid/-/uuid-3.4.0.tgz", - "integrity": "sha512-HjSDRw6gZE5JMggctHBcjVak08+KEVhSIiDzFnT9S9aegmp85S/bReBVTb4QTFaRNptJ9kuYaNhnbNEOkbKb/A==" - }, - "verror": { - "version": "1.10.0", - "resolved": "https://registry.npmjs.org/verror/-/verror-1.10.0.tgz", - "integrity": "sha1-OhBcoXBTr1XW4nDB+CiGguGNpAA=", - "requires": { - "assert-plus": "^1.0.0", - "core-util-is": "1.0.2", - "extsprintf": "^1.2.0" - } - } - } -} diff --git a/ghibli-gallery-crawler/scrape.go b/ghibli-gallery-crawler/scrape.go new file mode 100644 index 0000000..58e95a3 --- /dev/null +++ b/ghibli-gallery-crawler/scrape.go @@ -0,0 +1,92 @@ +package main + +import ( + "fmt" + "io" + "net/http" + "net/url" + + "golang.org/x/net/html" +) + +// scrapeImageURLs fetches a page and returns the absolute URLs of every image +// it references. +func scrapeImageURLs(client *http.Client, pageURL *url.URL) ([]string, error) { + resp, err := client.Get(pageURL.String()) + if err != nil { + return nil, fmt.Errorf("fetching %s: %w", pageURL, err) + } + defer resp.Body.Close() + + if resp.StatusCode != http.StatusOK { + return nil, fmt.Errorf("fetching %s: unexpected status %s", pageURL, resp.Status) + } + + return extractImageURLs(pageURL, resp.Body) +} + +// extractImageURLs pulls the src of every in the document, resolves it +// against base, and strips query strings so that cache-busting parameters do +// not end up in file names. Duplicates and non-absolute results are dropped. +func extractImageURLs(base *url.URL, r io.Reader) ([]string, error) { + doc, err := html.Parse(r) + if err != nil { + return nil, fmt.Errorf("parsing HTML: %w", err) + } + + var ( + urls []string + seen = make(map[string]bool) + walk func(*html.Node) + ) + + walk = func(n *html.Node) { + if n.Type == html.ElementNode && n.Data == "img" { + if src, ok := attr(n, "src"); ok { + if resolved, ok := resolveImageURL(base, src); ok && !seen[resolved] { + seen[resolved] = true + urls = append(urls, resolved) + } + } + } + for child := n.FirstChild; child != nil; child = child.NextSibling { + walk(child) + } + } + walk(doc) + + return urls, nil +} + +// attr returns the value of the named attribute, if the node has it. +func attr(n *html.Node, name string) (string, bool) { + for _, a := range n.Attr { + if a.Key == name { + return a.Val, true + } + } + return "", false +} + +// resolveImageURL makes src absolute relative to base and reports whether the +// result is a usable http(s) URL. +func resolveImageURL(base *url.URL, src string) (string, bool) { + ref, err := url.Parse(src) + if err != nil { + return "", false + } + + resolved := base.ResolveReference(ref) + // Drop the query so URLs like '/hsts-pixel.gif?c=3.2.5' yield clean names. + resolved.RawQuery = "" + resolved.Fragment = "" + + if resolved.Host == "" { + return "", false + } + // data: and other schemes are not downloadable files. + if resolved.Scheme != "http" && resolved.Scheme != "https" { + return "", false + } + return resolved.String(), true +}