feat: replace Python and JS downloaders with a Go crawler

Port both scripts into one Go program with a command per behavior:
gallery walks the numbered ghibli.jp film galleries, scrape downloads
every image referenced by a page.

Downloads now run through a bounded worker pool, and non-200 responses
are no longer written to disk as image files. Drop the orphaned
package-lock.json, which pinned the deprecated request dependency and
was the source of the repository's Dependabot alerts.
This commit is contained in:
tiennm99 committed 2026-07-25 19:41:43 +07:00
1 parent 07b6b84151
commit fb0964bb6e
13 files changed
+703 -492

No files matched your search

+11
View File
@@ -0,0 +1,11 @@
# Build output
/ghibli-gallery-crawler
/ghibli-gallery-crawler.exe
# Downloaded images
*.jpg
*.jpeg
*.png
*.gif
*.webp
*.svg
+50 -11
View File
@@ -1,27 +1,66 @@
# ghibli-gallery-crawler
Crawler for the Ghibli film gallery images (from 2020-08-31). Both Python and JavaScript variants included.
Image downloader in Go (originally written in Python and JavaScript in 2020-08-31).
## Install
```bash
go build -o ghibli-gallery-crawler .
```
Or run without building: `go run . <command>`.
## Usage
```bash
# Python
python3 download_images.py
Two commands, one per behavior carried over from the original scripts.
# JavaScript
npm install
node download_images.js
### `gallery` — numbered ghibli.jp galleries
Downloads `<film>001.jpg` … `<film>050.jpg` for each film, into one directory per film:
```bash
# All 8 default films into ./marnie, ./kaguyahime, ...
ghibli-gallery-crawler gallery
# Pick films, image count, and output location
ghibli-gallery-crawler gallery -films ponyo,chihiro -count 50 -out ./images
```
Edit the script to set the source URLs and target directory before running.
| Flag | Default | Meaning |
|---|---|---|
| `-films` | the 8 films below | comma-separated film slugs |
| `-count` | `50` | highest image number tried per film |
| `-out` | `.` | where the per-film directories are created |
| `-base` | `https://www.ghibli.jp/gallery/` | base gallery URL |
| `-concurrency` | `8` | parallel downloads |
| `-timeout` | `30s` | per-request timeout |
| `-user-agent` | tool identifier | `User-Agent` header to send |
## Input format
Default films: `marnie`, `kaguyahime`, `kazetachinu`, `kokurikozaka`, `karigurashi`, `ponyo`, `ged`, `chihiro`.
Both variants read a hardcoded list of image URLs defined at the top of each script. Edit the URL array in `download_images.py` or `download_images.js` to point at your images and set the output directory path.
Image numbers are a fixed range, so films with fewer than `-count` images are normal — those are reported as `missing`, not as failures.
### `scrape` — every image on a page
Parses a page's HTML, collects every `<img src>`, resolves relative URLs, and downloads them:
```bash
# Saves into ./en.wikipedia.org (the URL's host)
ghibli-gallery-crawler scrape https://en.wikipedia.org/wiki/Studio_Ghibli
# Explicit output directory
ghibli-gallery-crawler scrape https://example.com -path ./images
```
Flags: `-path` (default: the URL's host), plus `-concurrency`, `-timeout`, and `-user-agent` as above.
Query strings are stripped from image URLs so that names like `/hsts-pixel.gif?c=3.2.5` produce clean file names. `data:` URIs and duplicates are skipped.
## Output behavior
Images are saved into the configured output directory, named by their original filename from the URL. Existing files are overwritten. The Python variant uses `requests`; the JS variant uses `axios` + `fs.createWriteStream`.
Images are named after the last path segment of their URL. Existing files with the same name are overwritten. Missing directories are created. Non-200 responses are never written to disk, so error pages cannot land in your output as `.jpg` files.
Note that `ghibli.jp` serves gallery images to any client but returns `403` for its HTML pages unless the request looks like a browser, so `gallery` works there while `scrape` does not.
## License
+36
View File
@@ -0,0 +1,36 @@
package main
import (
"net/http"
"time"
)
// defaultUserAgent identifies this tool to servers. Some sites reject requests
// that do not send a User-Agent header at all, and an honest identifier lets
// operators see who is fetching. Override it with -user-agent.
const defaultUserAgent = "ghibli-gallery-crawler/1.0 (+https://github.com/tiennm99/ghibli-gallery-crawler)"
// userAgentTransport stamps every request with a User-Agent header.
type userAgentTransport struct {
userAgent string
base http.RoundTripper
}
func (t *userAgentTransport) RoundTrip(req *http.Request) (*http.Response, error) {
// Clone before mutating: RoundTrippers must not modify the caller's request.
clone := req.Clone(req.Context())
clone.Header.Set("User-Agent", t.userAgent)
return t.base.RoundTrip(clone)
}
// newClient builds an HTTP client that applies the given timeout to each
// request and identifies itself with userAgent.
func newClient(timeout time.Duration, userAgent string) *http.Client {
return &http.Client{
Timeout: timeout,
Transport: &userAgentTransport{
userAgent: userAgent,
base: http.DefaultTransport,
},
}
}
+112
View File
@@ -0,0 +1,112 @@
package main
import (
"net/url"
"path/filepath"
"strings"
"testing"
)
func TestExtractImageURLs(t *testing.T) {
base, err := url.Parse("https://example.com/gallery/index.html")
if err != nil {
t.Fatalf("parsing base: %v", err)
}
doc := `<html><body>
<img src="photo.jpg">
<img src="/absolute/other.png">
<img src="https://cdn.example.net/remote.gif">
<img src="/hsts-pixel.gif?c=3.2.5">
<img src="photo.jpg">
<img alt="no src attribute">
<img src="data:image/png;base64,AAAA">
</body></html>`
got, err := extractImageURLs(base, strings.NewReader(doc))
if err != nil {
t.Fatalf("extractImageURLs: %v", err)
}
want := []string{
"https://example.com/gallery/photo.jpg",
"https://example.com/absolute/other.png",
"https://cdn.example.net/remote.gif",
"https://example.com/hsts-pixel.gif", // query stripped
}
if len(got) != len(want) {
t.Fatalf("got %d URLs %v, want %d %v", len(got), got, len(want), want)
}
for i := range want {
if got[i] != want[i] {
t.Errorf("URL %d = %q, want %q", i, got[i], want[i])
}
}
}
func TestGalleryJobs(t *testing.T) {
jobs, err := galleryJobs("https://www.ghibli.jp/gallery/", []string{"ponyo", "ged"}, 3, "out")
if err != nil {
t.Fatalf("galleryJobs: %v", err)
}
if len(jobs) != 6 {
t.Fatalf("got %d jobs, want 6", len(jobs))
}
// Numbers are zero-padded to three digits, as in the original script.
if want := "https://www.ghibli.jp/gallery/ponyo001.jpg"; jobs[0].url != want {
t.Errorf("first URL = %q, want %q", jobs[0].url, want)
}
if want := filepath.Join("out", "ponyo"); jobs[0].dir != want {
t.Errorf("first dir = %q, want %q", jobs[0].dir, want)
}
if want := "https://www.ghibli.jp/gallery/ged003.jpg"; jobs[5].url != want {
t.Errorf("last URL = %q, want %q", jobs[5].url, want)
}
}
func TestFileNameFromURL(t *testing.T) {
cases := []struct {
in string
want string
wantErr bool
}{
{in: "https://example.com/a/b/chihiro001.jpg", want: "chihiro001.jpg"},
{in: "https://example.com/image.png", want: "image.png"},
{in: "https://example.com/", wantErr: true},
{in: "https://example.com", wantErr: true},
}
for _, c := range cases {
got, err := fileNameFromURL(c.in)
if c.wantErr {
if err == nil {
t.Errorf("fileNameFromURL(%q) = %q, want error", c.in, got)
}
continue
}
if err != nil {
t.Errorf("fileNameFromURL(%q): %v", c.in, err)
continue
}
if got != c.want {
t.Errorf("fileNameFromURL(%q) = %q, want %q", c.in, got, c.want)
}
}
}
func TestSplitFilms(t *testing.T) {
got := splitFilms(" ponyo , ,ged, ../escape ")
want := []string{"ponyo", "ged"}
if len(got) != len(want) {
t.Fatalf("got %v, want %v", got, want)
}
for i := range want {
if got[i] != want[i] {
t.Errorf("slug %d = %q, want %q", i, got[i], want[i])
}
}
}
+157
View File
@@ -0,0 +1,157 @@
package main
import (
"errors"
"fmt"
"io"
"net/http"
"net/url"
"os"
"path"
"path/filepath"
"sync"
)
// job is a single image to fetch and the directory to store it in.
type job struct {
url string
dir string
}
// errNotFound marks images the server does not have. The numbered gallery mode
// always asks for a fixed range, so missing images are expected, not failures.
var errNotFound = errors.New("not found")
// download fetches every job using at most concurrency parallel requests and
// reports how many succeeded. Individual failures are logged and do not stop
// the run.
func download(client *http.Client, jobs []job, concurrency int) error {
if concurrency < 1 {
concurrency = 1
}
if concurrency > len(jobs) {
concurrency = len(jobs)
}
// Create directories up front so parallel workers never race on MkdirAll.
for _, j := range jobs {
if err := os.MkdirAll(j.dir, 0o755); err != nil {
return fmt.Errorf("creating %s: %w", j.dir, err)
}
}
var (
queue = make(chan job)
wg sync.WaitGroup
mu sync.Mutex
saved int
missing int
failed int
)
for range concurrency {
wg.Add(1)
go func() {
defer wg.Done()
for j := range queue {
size, err := downloadOne(client, j)
mu.Lock()
switch {
case err == nil:
saved++
fmt.Printf("saved %s (%s)\n", j.url, humanSize(size))
case errors.Is(err, errNotFound):
missing++
default:
failed++
fmt.Fprintf(os.Stderr, "failed %s: %v\n", j.url, err)
}
mu.Unlock()
}
}()
}
for _, j := range jobs {
queue <- j
}
close(queue)
wg.Wait()
fmt.Printf("Done: %d saved, %d missing, %d failed\n", saved, missing, failed)
if saved == 0 && failed > 0 {
return fmt.Errorf("every download failed")
}
return nil
}
// downloadOne writes a single image to disk and returns the bytes written. An
// existing file with the same name is overwritten.
func downloadOne(client *http.Client, j job) (int64, error) {
name, err := fileNameFromURL(j.url)
if err != nil {
return 0, err
}
resp, err := client.Get(j.url)
if err != nil {
return 0, err
}
defer resp.Body.Close()
if resp.StatusCode == http.StatusNotFound || resp.StatusCode == http.StatusForbidden {
return 0, errNotFound
}
if resp.StatusCode != http.StatusOK {
// Bail out before writing, so error pages never land on disk as images.
return 0, fmt.Errorf("unexpected status %s", resp.Status)
}
dest := filepath.Join(j.dir, name)
f, err := os.Create(dest)
if err != nil {
return 0, err
}
defer f.Close()
written, err := io.Copy(f, resp.Body)
if err != nil {
// Leave no truncated file behind on a mid-transfer error.
os.Remove(dest)
return 0, err
}
return written, nil
}
// fileNameFromURL derives a safe local file name from the last path segment of
// an image URL.
func fileNameFromURL(rawURL string) (string, error) {
parsed, err := url.Parse(rawURL)
if err != nil {
return "", err
}
name := path.Base(parsed.Path)
// path.Base returns "." or "/" when there is nothing usable to take.
if name == "" || name == "." || name == "/" || name == ".." {
return "", fmt.Errorf("cannot derive a file name from %q", rawURL)
}
// Guard against a remote path smuggling in separators.
return filepath.Base(name), nil
}
// humanSize formats a byte count for display.
func humanSize(n int64) string {
const unit = 1024
if n < unit {
return fmt.Sprintf("%d B", n)
}
value := float64(n)
for _, suffix := range []string{"KiB", "MiB", "GiB"} {
value /= unit
if value < unit {
return fmt.Sprintf("%.1f %s", value, suffix)
}
}
return fmt.Sprintf("%.1f TiB", value)
}
-54
View File
@@ -1,54 +0,0 @@
const fs = require('fs')
const request = require('request')
const {
cookieCompare
} = require('tough-cookie')
const download = (url, path, callback) => {
request.head(url, (err, res, body) => {
request(url)
.pipe(fs.createWriteStream(path))
.on('close', callback)
})
}
// const url = 'https://…'
// const path = './images/image.png'
// download(url, path, () => {
// console.log('✅ Done!')
// })
// let urlConst = "https://www.iamag.co/wp-content/uploads/2017/07/yourname-background"
// for (let i = 1; i < 200; i++) {
// download(urlConst + i + ".jpeg", "img/" + i + ".jpeg", () => {
// console.log("Done: " + i);
// })
// }
function lpad(n, width, z) {
z = z || '0';
n = n + '';
return n.length >= width ? n : new Array(width - n.length + 1).join(z) + n;
}
let ghibli = "http://www.ghibli.jp/gallery/";
let listFilm = ["marnie", "kaguyahime", "kazetachinu", "kokurikozaka", "karigurashi", "ponyo", "ged", "chihiro"];
listFilm.forEach(film => {
let filmDir = "./" + film;
fs.mkdir(filmDir, function(err) {
if (err) {
console.log(err);
} else {
console.log("Created " + filmDir);
}
});
for (let i = 1; i <= 50; i++) {
let imgName = lpad(i, 3) + ".jpg";
let imgUrl = ghibli + film + imgName;
console.log(imgUrl);
download(imgUrl, filmDir + "/" + imgName, () => {
console.log("Done: " + imgName);
})
}
})
-90
View File
@@ -1,90 +0,0 @@
import requests
import os
from tqdm import tqdm
from bs4 import BeautifulSoup as bs
from urllib.parse import urljoin, urlparse
def is_valid(url):
"""
Checks whether `url` is a valid URL.
"""
parsed = urlparse(url)
return bool(parsed.netloc) and bool(parsed.scheme)
def get_all_images(url):
"""
Returns all image URLs on a single `url`
"""
soup = bs(requests.get(url).content, "html.parser")
urls = []
for img in tqdm(soup.find_all("img"), "Extracting images"):
img_url = img.attrs.get("src")
if not img_url:
# if img does not contain src attribute, just skip
continue
# make the URL absolute by joining domain with the URL that is just extracted
img_url = urljoin(url, img_url)
# remove URLs like '/hsts-pixel.gif?c=3.2.5'
try:
pos = img_url.index("?")
img_url = img_url[:pos]
except ValueError:
pass
# finally, if the url is valid
if is_valid(img_url):
urls.append(img_url)
return urls
def download(url, pathname):
"""
Downloads a file given an URL and puts it in the folder `pathname`
"""
# if path doesn't exist, make that path dir
if not os.path.isdir(pathname):
os.makedirs(pathname)
# download the body of response by chunk, not immediately
response = requests.get(url, stream=True)
# get the total file size
file_size = int(response.headers.get("Content-Length", 0))
# get the file name
filename = os.path.join(pathname, url.split("/")[-1])
# progress bar, changing the unit to bytes instead of iteration (default by tqdm)
progress = tqdm(response.iter_content(1024), f"Downloading {filename}", total=file_size, unit="B", unit_scale=True, unit_divisor=1024)
with open(filename, "wb") as f:
for data in progress:
# write data read to the file
f.write(data)
# update the progress bar manually
progress.update(len(data))
def main(url, path):
# get all images
imgs = get_all_images(url)
for img in imgs:
# for each img, download it
download(img, path)
if __name__ == "__main__":
import argparse
parser = argparse.ArgumentParser(description="This script downloads all images from a web page")
parser.add_argument("url", help="The URL of the web page you want to download images")
parser.add_argument("-p", "--path", help="The Directory you want to store your images, default is the domain of URL passed")
args = parser.parse_args()
url = args.url
path = args.path
if not path:
# if path isn't specified, use the domain name of that url as the folder name
path = urlparse(url).netloc
main(url, path)
+34
View File
@@ -0,0 +1,34 @@
package main
import (
"fmt"
"net/url"
"path/filepath"
)
// galleryJobs builds the download list for the numbered ghibli.jp galleries:
// for each film, images are named <film>001.jpg through <film><count>.jpg and
// are stored in a directory named after the film.
func galleryJobs(base string, films []string, count int, out string) ([]job, error) {
baseURL, err := url.Parse(base)
if err != nil {
return nil, fmt.Errorf("parsing base URL %q: %w", base, err)
}
if baseURL.Scheme == "" || baseURL.Host == "" {
return nil, fmt.Errorf("base URL %q must be absolute", base)
}
jobs := make([]job, 0, len(films)*count)
for _, film := range films {
dir := filepath.Join(out, film)
for i := 1; i <= count; i++ {
name := fmt.Sprintf("%s%03d.jpg", film, i)
imageURL, err := url.JoinPath(base, name)
if err != nil {
return nil, fmt.Errorf("building URL for %s: %w", name, err)
}
jobs = append(jobs, job{url: imageURL, dir: dir})
}
}
return jobs, nil
}
+5
View File
@@ -0,0 +1,5 @@
module github.com/tiennm99/ghibli-gallery-crawler
go 1.26.4
require golang.org/x/net v0.57.0
+2
View File
@@ -0,0 +1,2 @@
golang.org/x/net v0.57.0 h1:K5+3DljvIuDG9/Jv9rvyMywYNFCQ9RSUY6OOTTkT+tE=
golang.org/x/net v0.57.0/go.mod h1:KpXc8iv+r3XplLAG/f7Jsf9RPszJzdR0f58q9vGOuEU=
+204
View File
@@ -0,0 +1,204 @@
// Command ghibli-gallery-crawler downloads images from web pages.
//
// It offers two modes, mirroring the two scripts it replaces:
//
// scrape - parse a page's <img> tags and download every image found
// gallery - walk the numbered per-film galleries on ghibli.jp
package main
import (
"flag"
"fmt"
"net/url"
"os"
"path/filepath"
"strings"
"time"
)
const (
// defaultGalleryBase is where the numbered film galleries live. The
// original script used http://; https:// reaches the same files.
defaultGalleryBase = "https://www.ghibli.jp/gallery/"
// defaultGalleryCount is the highest image number tried per film.
defaultGalleryCount = 50
)
// defaultFilms lists the gallery slugs the original JavaScript script crawled.
var defaultFilms = []string{
"marnie", "kaguyahime", "kazetachinu", "kokurikozaka",
"karigurashi", "ponyo", "ged", "chihiro",
}
func main() {
if err := run(os.Args[1:]); err != nil {
fmt.Fprintln(os.Stderr, "error:", err)
os.Exit(1)
}
}
func run(args []string) error {
if len(args) == 0 {
usage()
return fmt.Errorf("no command given")
}
switch args[0] {
case "scrape":
return runScrape(args[1:])
case "gallery":
return runGallery(args[1:])
case "help", "-h", "--help":
usage()
return nil
default:
usage()
return fmt.Errorf("unknown command %q", args[0])
}
}
func usage() {
fmt.Fprint(os.Stderr, `ghibli-gallery-crawler downloads images from web pages.
Usage:
ghibli-gallery-crawler scrape <url> [flags]
ghibli-gallery-crawler gallery [flags]
Commands:
scrape Download every <img> found on a single page.
gallery Download the numbered ghibli.jp galleries for a list of films.
Run a command with -h to see its flags.
`)
}
// runScrape downloads every image referenced by a single page. The default
// output directory is the page's host, matching the Python script it replaces.
func runScrape(args []string) error {
fs := flag.NewFlagSet("scrape", flag.ExitOnError)
path := fs.String("path", "", "directory to store images in (default: the URL's host)")
concurrency := fs.Int("concurrency", 8, "number of parallel downloads")
timeout := fs.Duration("timeout", 30*time.Second, "per-request timeout")
userAgent := fs.String("user-agent", defaultUserAgent, "User-Agent header to send")
fs.Usage = func() {
fmt.Fprintln(fs.Output(), "Usage: ghibli-gallery-crawler scrape <url> [flags]")
fs.PrintDefaults()
}
positional, err := parseWithPositionals(fs, args)
if err != nil {
return err
}
if len(positional) != 1 {
fs.Usage()
return fmt.Errorf("scrape needs exactly one URL argument, got %d", len(positional))
}
pageURL := positional[0]
parsed, err := url.Parse(pageURL)
if err != nil {
return fmt.Errorf("parsing %q: %w", pageURL, err)
}
if parsed.Scheme == "" || parsed.Host == "" {
return fmt.Errorf("%q is not an absolute URL", pageURL)
}
dir := *path
if dir == "" {
dir = parsed.Host
}
client := newClient(*timeout, *userAgent)
imageURLs, err := scrapeImageURLs(client, parsed)
if err != nil {
return err
}
if len(imageURLs) == 0 {
fmt.Printf("No images found on %s\n", pageURL)
return nil
}
fmt.Printf("Found %d image(s) on %s\n", len(imageURLs), pageURL)
jobs := make([]job, 0, len(imageURLs))
for _, imageURL := range imageURLs {
jobs = append(jobs, job{url: imageURL, dir: dir})
}
return download(client, jobs, *concurrency)
}
// runGallery downloads images named <film>001.jpg .. <film>NNN.jpg for each
// film, storing each film's images in its own directory. This mirrors the
// JavaScript script it replaces.
func runGallery(args []string) error {
fs := flag.NewFlagSet("gallery", flag.ExitOnError)
base := fs.String("base", defaultGalleryBase, "base gallery URL")
films := fs.String("films", strings.Join(defaultFilms, ","), "comma-separated film slugs")
count := fs.Int("count", defaultGalleryCount, "highest image number to try per film")
out := fs.String("out", ".", "directory to create the per-film directories in")
concurrency := fs.Int("concurrency", 8, "number of parallel downloads")
timeout := fs.Duration("timeout", 30*time.Second, "per-request timeout")
userAgent := fs.String("user-agent", defaultUserAgent, "User-Agent header to send")
fs.Usage = func() {
fmt.Fprintln(fs.Output(), "Usage: ghibli-gallery-crawler gallery [flags]")
fs.PrintDefaults()
}
if err := fs.Parse(args); err != nil {
return err
}
if *count < 1 {
return fmt.Errorf("-count must be at least 1, got %d", *count)
}
slugs := splitFilms(*films)
if len(slugs) == 0 {
return fmt.Errorf("-films must name at least one film")
}
jobs, err := galleryJobs(*base, slugs, *count, *out)
if err != nil {
return err
}
fmt.Printf("Trying %d image(s) across %d film(s)\n", len(jobs), len(slugs))
client := newClient(*timeout, *userAgent)
return download(client, jobs, *concurrency)
}
// parseWithPositionals parses flags that may appear before or after positional
// arguments. The standard flag package stops at the first non-flag argument, so
// the remainder is fed back through the parser one positional at a time.
func parseWithPositionals(fs *flag.FlagSet, args []string) ([]string, error) {
var positional []string
for len(args) > 0 {
if err := fs.Parse(args); err != nil {
return nil, err
}
args = fs.Args()
if len(args) > 0 {
positional = append(positional, args[0])
args = args[1:]
}
}
return positional, nil
}
// splitFilms turns a comma-separated flag value into clean slugs, dropping
// empty entries and anything that would escape the output directory.
func splitFilms(value string) []string {
var slugs []string
for _, raw := range strings.Split(value, ",") {
slug := strings.TrimSpace(raw)
if slug == "" {
continue
}
// A slug becomes a directory name, so reject path separators.
if slug != filepath.Base(slug) || slug == "." || slug == ".." {
fmt.Fprintf(os.Stderr, "skipping invalid film slug %q\n", slug)
continue
}
slugs = append(slugs, slug)
}
return slugs
}
-337
View File
@@ -1,337 +0,0 @@
{
"requires": true,
"lockfileVersion": 1,
"dependencies": {
"ajv": {
"version": "6.12.4",
"resolved": "https://registry.npmjs.org/ajv/-/ajv-6.12.4.tgz",
"integrity": "sha512-eienB2c9qVQs2KWexhkrdMLVDoIQCz5KSeLxwg9Lzk4DOfBtIK9PQwwufcsn1jjGuf9WZmqPMbGxOzfcuphJCQ==",
"requires": {
"fast-deep-equal": "^3.1.1",
"fast-json-stable-stringify": "^2.0.0",
"json-schema-traverse": "^0.4.1",
"uri-js": "^4.2.2"
}
},
"asn1": {
"version": "0.2.4",
"resolved": "https://registry.npmjs.org/asn1/-/asn1-0.2.4.tgz",
"integrity": "sha512-jxwzQpLQjSmWXgwaCZE9Nz+glAG01yF1QnWgbhGwHI5A6FRIEY6IVqtHhIepHqI7/kyEyQEagBC5mBEFlIYvdg==",
"requires": {
"safer-buffer": "~2.1.0"
}
},
"assert-plus": {
"version": "1.0.0",
"resolved": "https://registry.npmjs.org/assert-plus/-/assert-plus-1.0.0.tgz",
"integrity": "sha1-8S4PPF13sLHN2RRpQuTpbB5N1SU="
},
"asynckit": {
"version": "0.4.0",
"resolved": "https://registry.npmjs.org/asynckit/-/asynckit-0.4.0.tgz",
"integrity": "sha1-x57Zf380y48robyXkLzDZkdLS3k="
},
"aws-sign2": {
"version": "0.7.0",
"resolved": "https://registry.npmjs.org/aws-sign2/-/aws-sign2-0.7.0.tgz",
"integrity": "sha1-tG6JCTSpWR8tL2+G1+ap8bP+dqg="
},
"aws4": {
"version": "1.10.1",
"resolved": "https://registry.npmjs.org/aws4/-/aws4-1.10.1.tgz",
"integrity": "sha512-zg7Hz2k5lI8kb7U32998pRRFin7zJlkfezGJjUc2heaD4Pw2wObakCDVzkKztTm/Ln7eiVvYsjqak0Ed4LkMDA=="
},
"bcrypt-pbkdf": {
"version": "1.0.2",
"resolved": "https://registry.npmjs.org/bcrypt-pbkdf/-/bcrypt-pbkdf-1.0.2.tgz",
"integrity": "sha1-pDAdOJtqQ/m2f/PKEaP2Y342Dp4=",
"requires": {
"tweetnacl": "^0.14.3"
}
},
"caseless": {
"version": "0.12.0",
"resolved": "https://registry.npmjs.org/caseless/-/caseless-0.12.0.tgz",
"integrity": "sha1-G2gcIf+EAzyCZUMJBolCDRhxUdw="
},
"combined-stream": {
"version": "1.0.8",
"resolved": "https://registry.npmjs.org/combined-stream/-/combined-stream-1.0.8.tgz",
"integrity": "sha512-FQN4MRfuJeHf7cBbBMJFXhKSDq+2kAArBlmRBvcvFE5BB1HZKXtSFASDhdlz9zOYwxh8lDdnvmMOe/+5cdoEdg==",
"requires": {
"delayed-stream": "~1.0.0"
}
},
"core-util-is": {
"version": "1.0.2",
"resolved": "https://registry.npmjs.org/core-util-is/-/core-util-is-1.0.2.tgz",
"integrity": "sha1-tf1UIgqivFq1eqtxQMlAdUUDwac="
},
"dashdash": {
"version": "1.14.1",
"resolved": "https://registry.npmjs.org/dashdash/-/dashdash-1.14.1.tgz",
"integrity": "sha1-hTz6D3y+L+1d4gMmuN1YEDX24vA=",
"requires": {
"assert-plus": "^1.0.0"
}
},
"delayed-stream": {
"version": "1.0.0",
"resolved": "https://registry.npmjs.org/delayed-stream/-/delayed-stream-1.0.0.tgz",
"integrity": "sha1-3zrhmayt+31ECqrgsp4icrJOxhk="
},
"ecc-jsbn": {
"version": "0.1.2",
"resolved": "https://registry.npmjs.org/ecc-jsbn/-/ecc-jsbn-0.1.2.tgz",
"integrity": "sha1-OoOpBOVDUyh4dMVkt1SThoSamMk=",
"requires": {
"jsbn": "~0.1.0",
"safer-buffer": "^2.1.0"
}
},
"extend": {
"version": "3.0.2",
"resolved": "https://registry.npmjs.org/extend/-/extend-3.0.2.tgz",
"integrity": "sha512-fjquC59cD7CyW6urNXK0FBufkZcoiGG80wTuPujX590cB5Ttln20E2UB4S/WARVqhXffZl2LNgS+gQdPIIim/g=="
},
"extsprintf": {
"version": "1.3.0",
"resolved": "https://registry.npmjs.org/extsprintf/-/extsprintf-1.3.0.tgz",
"integrity": "sha1-lpGEQOMEGnpBT4xS48V06zw+HgU="
},
"fast-deep-equal": {
"version": "3.1.3",
"resolved": "https://registry.npmjs.org/fast-deep-equal/-/fast-deep-equal-3.1.3.tgz",
"integrity": "sha512-f3qQ9oQy9j2AhBe/H9VC91wLmKBCCU/gDOnKNAYG5hswO7BLKj09Hc5HYNz9cGI++xlpDCIgDaitVs03ATR84Q=="
},
"fast-json-stable-stringify": {
"version": "2.1.0",
"resolved": "https://registry.npmjs.org/fast-json-stable-stringify/-/fast-json-stable-stringify-2.1.0.tgz",
"integrity": "sha512-lhd/wF+Lk98HZoTCtlVraHtfh5XYijIjalXck7saUtuanSDyLMxnHhSXEDJqHxD7msR8D0uCmqlkwjCV8xvwHw=="
},
"forever-agent": {
"version": "0.6.1",
"resolved": "https://registry.npmjs.org/forever-agent/-/forever-agent-0.6.1.tgz",
"integrity": "sha1-+8cfDEGt6zf5bFd60e1C2P2sypE="
},
"form-data": {
"version": "2.3.3",
"resolved": "https://registry.npmjs.org/form-data/-/form-data-2.3.3.tgz",
"integrity": "sha512-1lLKB2Mu3aGP1Q/2eCOx0fNbRMe7XdwktwOruhfqqd0rIJWwN4Dh+E3hrPSlDCXnSR7UtZ1N38rVXm+6+MEhJQ==",
"requires": {
"asynckit": "^0.4.0",
"combined-stream": "^1.0.6",
"mime-types": "^2.1.12"
}
},
"getpass": {
"version": "0.1.7",
"resolved": "https://registry.npmjs.org/getpass/-/getpass-0.1.7.tgz",
"integrity": "sha1-Xv+OPmhNVprkyysSgmBOi6YhSfo=",
"requires": {
"assert-plus": "^1.0.0"
}
},
"har-schema": {
"version": "2.0.0",
"resolved": "https://registry.npmjs.org/har-schema/-/har-schema-2.0.0.tgz",
"integrity": "sha1-qUwiJOvKwEeCoNkDVSHyRzW37JI="
},
"har-validator": {
"version": "5.1.5",
"resolved": "https://registry.npmjs.org/har-validator/-/har-validator-5.1.5.tgz",
"integrity": "sha512-nmT2T0lljbxdQZfspsno9hgrG3Uir6Ks5afism62poxqBM6sDnMEuPmzTq8XN0OEwqKLLdh1jQI3qyE66Nzb3w==",
"requires": {
"ajv": "^6.12.3",
"har-schema": "^2.0.0"
}
},
"http-signature": {
"version": "1.2.0",
"resolved": "https://registry.npmjs.org/http-signature/-/http-signature-1.2.0.tgz",
"integrity": "sha1-muzZJRFHcvPZW2WmCruPfBj7rOE=",
"requires": {
"assert-plus": "^1.0.0",
"jsprim": "^1.2.2",
"sshpk": "^1.7.0"
}
},
"is-typedarray": {
"version": "1.0.0",
"resolved": "https://registry.npmjs.org/is-typedarray/-/is-typedarray-1.0.0.tgz",
"integrity": "sha1-5HnICFjfDBsR3dppQPlgEfzaSpo="
},
"isstream": {
"version": "0.1.2",
"resolved": "https://registry.npmjs.org/isstream/-/isstream-0.1.2.tgz",
"integrity": "sha1-R+Y/evVa+m+S4VAOaQ64uFKcCZo="
},
"jsbn": {
"version": "0.1.1",
"resolved": "https://registry.npmjs.org/jsbn/-/jsbn-0.1.1.tgz",
"integrity": "sha1-peZUwuWi3rXyAdls77yoDA7y9RM="
},
"json-schema": {
"version": "0.2.3",
"resolved": "https://registry.npmjs.org/json-schema/-/json-schema-0.2.3.tgz",
"integrity": "sha1-tIDIkuWaLwWVTOcnvT8qTogvnhM="
},
"json-schema-traverse": {
"version": "0.4.1",
"resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-0.4.1.tgz",
"integrity": "sha512-xbbCH5dCYU5T8LcEhhuh7HJ88HXuW3qsI3Y0zOZFKfZEHcpWiHU/Jxzk629Brsab/mMiHQti9wMP+845RPe3Vg=="
},
"json-stringify-safe": {
"version": "5.0.1",
"resolved": "https://registry.npmjs.org/json-stringify-safe/-/json-stringify-safe-5.0.1.tgz",
"integrity": "sha1-Epai1Y/UXxmg9s4B1lcB4sc1tus="
},
"jsprim": {
"version": "1.4.1",
"resolved": "https://registry.npmjs.org/jsprim/-/jsprim-1.4.1.tgz",
"integrity": "sha1-MT5mvB5cwG5Di8G3SZwuXFastqI=",
"requires": {
"assert-plus": "1.0.0",
"extsprintf": "1.3.0",
"json-schema": "0.2.3",
"verror": "1.10.0"
}
},
"mime-db": {
"version": "1.44.0",
"resolved": "https://registry.npmjs.org/mime-db/-/mime-db-1.44.0.tgz",
"integrity": "sha512-/NOTfLrsPBVeH7YtFPgsVWveuL+4SjjYxaQ1xtM1KMFj7HdxlBlxeyNLzhyJVx7r4rZGJAZ/6lkKCitSc/Nmpg=="
},
"mime-types": {
"version": "2.1.27",
"resolved": "https://registry.npmjs.org/mime-types/-/mime-types-2.1.27.tgz",
"integrity": "sha512-JIhqnCasI9yD+SsmkquHBxTSEuZdQX5BuQnS2Vc7puQQQ+8yiP5AY5uWhpdv4YL4VM5c6iliiYWPgJ/nJQLp7w==",
"requires": {
"mime-db": "1.44.0"
}
},
"oauth-sign": {
"version": "0.9.0",
"resolved": "https://registry.npmjs.org/oauth-sign/-/oauth-sign-0.9.0.tgz",
"integrity": "sha512-fexhUFFPTGV8ybAtSIGbV6gOkSv8UtRbDBnAyLQw4QPKkgNlsH2ByPGtMUqdWkos6YCRmAqViwgZrJc/mRDzZQ=="
},
"performance-now": {
"version": "2.1.0",
"resolved": "https://registry.npmjs.org/performance-now/-/performance-now-2.1.0.tgz",
"integrity": "sha1-Ywn04OX6kT7BxpMHrjZLSzd8nns="
},
"psl": {
"version": "1.8.0",
"resolved": "https://registry.npmjs.org/psl/-/psl-1.8.0.tgz",
"integrity": "sha512-RIdOzyoavK+hA18OGGWDqUTsCLhtA7IcZ/6NCs4fFJaHBDab+pDDmDIByWFRQJq2Cd7r1OoQxBGKOaztq+hjIQ=="
},
"punycode": {
"version": "2.1.1",
"resolved": "https://registry.npmjs.org/punycode/-/punycode-2.1.1.tgz",
"integrity": "sha512-XRsRjdf+j5ml+y/6GKHPZbrF/8p2Yga0JPtdqTIY2Xe5ohJPD9saDJJLPvp9+NSBprVvevdXZybnj2cv8OEd0A=="
},
"qs": {
"version": "6.5.2",
"resolved": "https://registry.npmjs.org/qs/-/qs-6.5.2.tgz",
"integrity": "sha512-N5ZAX4/LxJmF+7wN74pUD6qAh9/wnvdQcjq9TZjevvXzSUo7bfmw91saqMjzGS2xq91/odN2dW/WOl7qQHNDGA=="
},
"request": {
"version": "2.88.2",
"resolved": "https://registry.npmjs.org/request/-/request-2.88.2.tgz",
"integrity": "sha512-MsvtOrfG9ZcrOwAW+Qi+F6HbD0CWXEh9ou77uOb7FM2WPhwT7smM833PzanhJLsgXjN89Ir6V2PczXNnMpwKhw==",
"requires": {
"aws-sign2": "~0.7.0",
"aws4": "^1.8.0",
"caseless": "~0.12.0",
"combined-stream": "~1.0.6",
"extend": "~3.0.2",
"forever-agent": "~0.6.1",
"form-data": "~2.3.2",
"har-validator": "~5.1.3",
"http-signature": "~1.2.0",
"is-typedarray": "~1.0.0",
"isstream": "~0.1.2",
"json-stringify-safe": "~5.0.1",
"mime-types": "~2.1.19",
"oauth-sign": "~0.9.0",
"performance-now": "^2.1.0",
"qs": "~6.5.2",
"safe-buffer": "^5.1.2",
"tough-cookie": "~2.5.0",
"tunnel-agent": "^0.6.0",
"uuid": "^3.3.2"
}
},
"safe-buffer": {
"version": "5.2.1",
"resolved": "https://registry.npmjs.org/safe-buffer/-/safe-buffer-5.2.1.tgz",
"integrity": "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ=="
},
"safer-buffer": {
"version": "2.1.2",
"resolved": "https://registry.npmjs.org/safer-buffer/-/safer-buffer-2.1.2.tgz",
"integrity": "sha512-YZo3K82SD7Riyi0E1EQPojLz7kpepnSQI9IyPbHHg1XXXevb5dJI7tpyN2ADxGcQbHG7vcyRHk0cbwqcQriUtg=="
},
"sshpk": {
"version": "1.16.1",
"resolved": "https://registry.npmjs.org/sshpk/-/sshpk-1.16.1.tgz",
"integrity": "sha512-HXXqVUq7+pcKeLqqZj6mHFUMvXtOJt1uoUx09pFW6011inTMxqI8BA8PM95myrIyyKwdnzjdFjLiE6KBPVtJIg==",
"requires": {
"asn1": "~0.2.3",
"assert-plus": "^1.0.0",
"bcrypt-pbkdf": "^1.0.0",
"dashdash": "^1.12.0",
"ecc-jsbn": "~0.1.1",
"getpass": "^0.1.1",
"jsbn": "~0.1.0",
"safer-buffer": "^2.0.2",
"tweetnacl": "~0.14.0"
}
},
"tough-cookie": {
"version": "2.5.0",
"resolved": "https://registry.npmjs.org/tough-cookie/-/tough-cookie-2.5.0.tgz",
"integrity": "sha512-nlLsUzgm1kfLXSXfRZMc1KLAugd4hqJHDTvc2hDIwS3mZAfMEuMbc03SujMF+GEcpaX/qboeycw6iO8JwVv2+g==",
"requires": {
"psl": "^1.1.28",
"punycode": "^2.1.1"
}
},
"tunnel-agent": {
"version": "0.6.0",
"resolved": "https://registry.npmjs.org/tunnel-agent/-/tunnel-agent-0.6.0.tgz",
"integrity": "sha1-J6XeoGs2sEoKmWZ3SykIaPD8QP0=",
"requires": {
"safe-buffer": "^5.0.1"
}
},
"tweetnacl": {
"version": "0.14.5",
"resolved": "https://registry.npmjs.org/tweetnacl/-/tweetnacl-0.14.5.tgz",
"integrity": "sha1-WuaBd/GS1EViadEIr6k/+HQ/T2Q="
},
"uri-js": {
"version": "4.4.0",
"resolved": "https://registry.npmjs.org/uri-js/-/uri-js-4.4.0.tgz",
"integrity": "sha512-B0yRTzYdUCCn9n+F4+Gh4yIDtMQcaJsmYBDsTSG8g/OejKBodLQ2IHfN3bM7jUsRXndopT7OIXWdYqc1fjmV6g==",
"requires": {
"punycode": "^2.1.0"
}
},
"uuid": {
"version": "3.4.0",
"resolved": "https://registry.npmjs.org/uuid/-/uuid-3.4.0.tgz",
"integrity": "sha512-HjSDRw6gZE5JMggctHBcjVak08+KEVhSIiDzFnT9S9aegmp85S/bReBVTb4QTFaRNptJ9kuYaNhnbNEOkbKb/A=="
},
"verror": {
"version": "1.10.0",
"resolved": "https://registry.npmjs.org/verror/-/verror-1.10.0.tgz",
"integrity": "sha1-OhBcoXBTr1XW4nDB+CiGguGNpAA=",
"requires": {
"assert-plus": "^1.0.0",
"core-util-is": "1.0.2",
"extsprintf": "^1.2.0"
}
}
}
}
+92
View File
@@ -0,0 +1,92 @@
package main
import (
"fmt"
"io"
"net/http"
"net/url"
"golang.org/x/net/html"
)
// scrapeImageURLs fetches a page and returns the absolute URLs of every image
// it references.
func scrapeImageURLs(client *http.Client, pageURL *url.URL) ([]string, error) {
resp, err := client.Get(pageURL.String())
if err != nil {
return nil, fmt.Errorf("fetching %s: %w", pageURL, err)
}
defer resp.Body.Close()
if resp.StatusCode != http.StatusOK {
return nil, fmt.Errorf("fetching %s: unexpected status %s", pageURL, resp.Status)
}
return extractImageURLs(pageURL, resp.Body)
}
// extractImageURLs pulls the src of every <img> in the document, resolves it
// against base, and strips query strings so that cache-busting parameters do
// not end up in file names. Duplicates and non-absolute results are dropped.
func extractImageURLs(base *url.URL, r io.Reader) ([]string, error) {
doc, err := html.Parse(r)
if err != nil {
return nil, fmt.Errorf("parsing HTML: %w", err)
}
var (
urls []string
seen = make(map[string]bool)
walk func(*html.Node)
)
walk = func(n *html.Node) {
if n.Type == html.ElementNode && n.Data == "img" {
if src, ok := attr(n, "src"); ok {
if resolved, ok := resolveImageURL(base, src); ok && !seen[resolved] {
seen[resolved] = true
urls = append(urls, resolved)
}
}
}
for child := n.FirstChild; child != nil; child = child.NextSibling {
walk(child)
}
}
walk(doc)
return urls, nil
}
// attr returns the value of the named attribute, if the node has it.
func attr(n *html.Node, name string) (string, bool) {
for _, a := range n.Attr {
if a.Key == name {
return a.Val, true
}
}
return "", false
}
// resolveImageURL makes src absolute relative to base and reports whether the
// result is a usable http(s) URL.
func resolveImageURL(base *url.URL, src string) (string, bool) {
ref, err := url.Parse(src)
if err != nil {
return "", false
}
resolved := base.ResolveReference(ref)
// Drop the query so URLs like '/hsts-pixel.gif?c=3.2.5' yield clean names.
resolved.RawQuery = ""
resolved.Fragment = ""
if resolved.Host == "" {
return "", false
}
// data: and other schemes are not downloadable files.
if resolved.Scheme != "http" && resolved.Scheme != "https" {
return "", false
}
return resolved.String(), true
}