From 3ebdc5dfbda8fd4a926a5c8c0aa3793fb46f5c7a Mon Sep 17 00:00:00 2001 From: Neo <321592+Neo-Desktop@users.noreply.github.com> Date: Fri, 31 Mar 2023 01:58:51 -0700 Subject: [PATCH] add function to scrape the upper bound of the search pagination before running main loop add compiling instructions to readme update .gitignore --- .gitignore | 3 +++ .idea/golinter.xml | 6 ++++++ README.md | 12 ++++++++++++ main.go | 47 +++++++++++++++++++++++++++++++++++++++++++--- 4 files changed, 65 insertions(+), 3 deletions(-) create mode 100644 .idea/golinter.xml diff --git a/.gitignore b/.gitignore index 7a11cb3..18c65c0 100644 --- a/.gitignore +++ b/.gitignore @@ -1,5 +1,8 @@ *.csv *.log +.backups +builds/ +winworldpc-ipfs-scraper ### NotepadPP template # Notepad++ backups # diff --git a/.idea/golinter.xml b/.idea/golinter.xml new file mode 100644 index 0000000..90da4a7 --- /dev/null +++ b/.idea/golinter.xml @@ -0,0 +1,6 @@ + + + + + \ No newline at end of file diff --git a/README.md b/README.md index b00514d..e3c602b 100644 --- a/README.md +++ b/README.md @@ -1,2 +1,14 @@ # winworldpc-ipfs-scraper grabs ipfs links for files from winworldpc + +------------ + +no downloads are provided at this time + +simply clone the project and run `go build` + +running the resulting executable will begin the scraping process + +progress is not saved and will restart from the first page, while appending to existing files + +hopefully you found this project useful diff --git a/main.go b/main.go index 26b5112..4ce88a9 100644 --- a/main.go +++ b/main.go @@ -5,9 +5,11 @@ import ( "fmt" "io" "log" + "math/bits" "net/http" "net/url" "os" + "strconv" "strings" "time" @@ -19,7 +21,7 @@ const ( csvFileName = "results.csv" csvFullResultsFileName = "results_full.csv" logFileName = "output.log" - UserAgent = "Mozilla/5.0 (compatible; IPFS.ScraperBot/v1.0; +https://github.com/Neo-Desktop/winworldpc-ipfs-scraper)" + UserAgent = "Mozilla/5.0 (compatible; IPFS.ScraperBot/v1.1; +https://github.com/Neo-Desktop/winworldpc-ipfs-scraper)" ) type File struct { @@ -57,7 +59,43 @@ type Article struct { Files []File } -func scrapeSearchPage(page int) []Article { +func scrapeSearchPageForUpperBound() uint { + urlA, err := url.Parse(URL + "/search") + if err != nil { + log.Fatal(err) + } + + log.Printf("Fetching search paginaion upper bound") + + res, err := fetch(urlA.String()) + if err != nil { + log.Fatal(err) + } + + defer res.Body.Close() + if res.StatusCode != http.StatusOK { + log.Fatalf("status code error: %d %s", res.StatusCode, res.Status) + } + + doc, err := gq.NewDocumentFromReader(res.Body) + if err != nil { + log.Fatal(err) + } + + lastPage := doc.Find("#searchPagination>li").Last() + lastPageText := strings.TrimSpace(lastPage.Text()) + + out, err := strconv.ParseUint(lastPageText, 10, bits.UintSize) + if err != nil { + log.Fatalf("unable to parse search pagination upper bound") + } + + log.Printf("====== Found %d total pages ======", out) + + return uint(out) +} + +func scrapeSearchPage(page uint) []Article { log.Printf("=============================== PAGE %2d ===============================", page) urlA, err := url.Parse(URL + "/search") if err != nil { @@ -65,6 +103,7 @@ func scrapeSearchPage(page int) []Article { } queryParameters := urlA.Query() + queryParameters.Add("sort", "most-recent") queryParameters.Add("page", fmt.Sprintf("%d", page)) urlA.RawQuery = queryParameters.Encode() @@ -253,7 +292,9 @@ func main() { articles := make([]Article, 0) - for i := 1; i < 63; i++ { + pageUpperBound := scrapeSearchPageForUpperBound() + + for i := uint(1); i < pageUpperBound; i++ { results := scrapeSearchPage(i) articles = append(articles, results...) go writeCSV(results, csvFileName)