package main import ( "fmt" "io" "net/http" "net/url" "strings" "time" ) // fetch makes a GET request to refURL, returning the HTML contents of // that webpage. An error is also returned. // // A [url.URL] type is used for refURL to simplify recursive or else // repeated use of this function when crawling webpages to, say, build // a sitemap. func fetch(refURL url.URL) ([]byte, error) { rawURL := refURL.String() // FIXME: make the timeout configurable. client := http.Client{ Timeout: 2 * time.Second, } req, err := http.NewRequest(http.MethodGet, rawURL, nil) if err != nil { return nil, fmt.Errorf("can't create request: %w", err) } req.Header.Add("user-agent", "urls/1.0, GNU/Linux") resp, err := client.Do(req) if err != nil { return nil, fmt.Errorf("client failed: %w", err) } defer resp.Body.Close() if resp.StatusCode != http.StatusOK { return nil, fmt.Errorf("status for %s for %s: %s", http.MethodGet, rawURL, resp.Status) } if contentType := resp.Header.Get("content-type"); !goodContentType(contentType) { return nil, fmt.Errorf("non-html content-type: %s", contentType) } htmlBytes, err := io.ReadAll(resp.Body) if err != nil { return nil, fmt.Errorf("can't read reponse body into byte buffer") } return htmlBytes, nil } func goodContentType(contentType string) bool { what := strings.TrimSpace(strings.SplitN(contentType, ";", 2)[0]) return what == "text/html" }