add function to scrape the upper bound of the search pagination before running main loop

add compiling instructions to readme
update .gitignore
This commit is contained in:
Neo committed 2023-03-31 01:58:51 -07:00
1 parent 1328f837ee
commit 3ebdc5dfbd
4 files changed
+65 -3

No files matched your search

+3
View File
@@ -1,5 +1,8 @@
*.csv
*.log
.backups
builds/
winworldpc-ipfs-scraper
### NotepadPP template
# Notepad++ backups #
+6
View File
@@ -0,0 +1,6 @@
<?xml version="1.0" encoding="UTF-8"?>
<project version="4">
<component name="GoLinterSettings">
<option name="checkGoLinterExe" value="false" />
</component>
</project>
+12
View File
@@ -1,2 +1,14 @@
# winworldpc-ipfs-scraper
grabs ipfs links for files from winworldpc
------------
no downloads are provided at this time
simply clone the project and run `go build`
running the resulting executable will begin the scraping process
progress is not saved and will restart from the first page, while appending to existing files
hopefully you found this project useful
+44 -3
View File
@@ -5,9 +5,11 @@ import (
"fmt"
"io"
"log"
"math/bits"
"net/http"
"net/url"
"os"
"strconv"
"strings"
"time"
@@ -19,7 +21,7 @@ const (
csvFileName = "results.csv"
csvFullResultsFileName = "results_full.csv"
logFileName = "output.log"
UserAgent = "Mozilla/5.0 (compatible; IPFS.ScraperBot/v1.0; +https://github.com/Neo-Desktop/winworldpc-ipfs-scraper)"
UserAgent = "Mozilla/5.0 (compatible; IPFS.ScraperBot/v1.1; +https://github.com/Neo-Desktop/winworldpc-ipfs-scraper)"
)
type File struct {
@@ -57,7 +59,43 @@ type Article struct {
Files []File
}
func scrapeSearchPage(page int) []Article {
func scrapeSearchPageForUpperBound() uint {
urlA, err := url.Parse(URL + "/search")
if err != nil {
log.Fatal(err)
}
log.Printf("Fetching search paginaion upper bound")
res, err := fetch(urlA.String())
if err != nil {
log.Fatal(err)
}
defer res.Body.Close()
if res.StatusCode != http.StatusOK {
log.Fatalf("status code error: %d %s", res.StatusCode, res.Status)
}
doc, err := gq.NewDocumentFromReader(res.Body)
if err != nil {
log.Fatal(err)
}
lastPage := doc.Find("#searchPagination>li").Last()
lastPageText := strings.TrimSpace(lastPage.Text())
out, err := strconv.ParseUint(lastPageText, 10, bits.UintSize)
if err != nil {
log.Fatalf("unable to parse search pagination upper bound")
}
log.Printf("====== Found %d total pages ======", out)
return uint(out)
}
func scrapeSearchPage(page uint) []Article {
log.Printf("=============================== PAGE %2d ===============================", page)
urlA, err := url.Parse(URL + "/search")
if err != nil {
@@ -65,6 +103,7 @@ func scrapeSearchPage(page int) []Article {
}
queryParameters := urlA.Query()
queryParameters.Add("sort", "most-recent")
queryParameters.Add("page", fmt.Sprintf("%d", page))
urlA.RawQuery = queryParameters.Encode()
@@ -253,7 +292,9 @@ func main() {
articles := make([]Article, 0)
for i := 1; i < 63; i++ {
pageUpperBound := scrapeSearchPageForUpperBound()
for i := uint(1); i < pageUpperBound; i++ {
results := scrapeSearchPage(i)
articles = append(articles, results...)
go writeCSV(results, csvFileName)