From fc36b0eaae498d209c9d31e3f73757675fe1f3ca Mon Sep 17 00:00:00 2001
From: Neo <321592+Neo-Desktop@users.noreply.github.com>
Date: Sat, 28 Jan 2023 10:27:43 -0800
Subject: [PATCH] initial commit, working code
---
.gitignore | 225 +++++++++++++++++++++++++
.idea/.gitignore | 8 +
.idea/csv-editor.xml | 23 +++
.idea/modules.xml | 8 +
.idea/vcs.xml | 6 +
.idea/winworldpc-ipfs-scraper.iml | 9 +
go.mod | 10 ++
go.sum | 11 ++
main.go | 269 ++++++++++++++++++++++++++++++
9 files changed, 569 insertions(+)
create mode 100644 .gitignore
create mode 100644 .idea/.gitignore
create mode 100644 .idea/csv-editor.xml
create mode 100644 .idea/modules.xml
create mode 100644 .idea/vcs.xml
create mode 100644 .idea/winworldpc-ipfs-scraper.iml
create mode 100644 go.mod
create mode 100644 go.sum
create mode 100644 main.go
diff --git a/.gitignore b/.gitignore
new file mode 100644
index 0000000..dd3b96b
--- /dev/null
+++ b/.gitignore
@@ -0,0 +1,225 @@
+*.csv
+*.log
+
+### NotepadPP template
+# Notepad++ backups #
+*.bak
+
+### VisualStudioCode template
+.vscode/*
+!.vscode/settings.json
+!.vscode/tasks.json
+!.vscode/launch.json
+!.vscode/extensions.json
+!.vscode/*.code-snippets
+
+# Local History for Visual Studio Code
+.history/
+
+# Built Visual Studio Code Extensions
+*.vsix
+
+### JetBrains template
+# Covers JetBrains IDEs: IntelliJ, RubyMine, PhpStorm, AppCode, PyCharm, CLion, Android Studio, WebStorm and Rider
+# Reference: https://intellij-support.jetbrains.com/hc/en-us/articles/206544839
+
+# User-specific stuff
+.idea/**/workspace.xml
+.idea/**/tasks.xml
+.idea/**/usage.statistics.xml
+.idea/**/dictionaries
+.idea/**/shelf
+
+# AWS User-specific
+.idea/**/aws.xml
+
+# Generated files
+.idea/**/contentModel.xml
+
+# Sensitive or high-churn files
+.idea/**/dataSources/
+.idea/**/dataSources.ids
+.idea/**/dataSources.local.xml
+.idea/**/sqlDataSources.xml
+.idea/**/dynamic.xml
+.idea/**/uiDesigner.xml
+.idea/**/dbnavigator.xml
+
+# Gradle
+.idea/**/gradle.xml
+.idea/**/libraries
+
+# Gradle and Maven with auto-import
+# When using Gradle or Maven with auto-import, you should exclude module files,
+# since they will be recreated, and may cause churn. Uncomment if using
+# auto-import.
+# .idea/artifacts
+# .idea/compiler.xml
+# .idea/jarRepositories.xml
+# .idea/modules.xml
+# .idea/*.iml
+# .idea/modules
+# *.iml
+# *.ipr
+
+# CMake
+cmake-build-*/
+
+# Mongo Explorer plugin
+.idea/**/mongoSettings.xml
+
+# File-based project format
+*.iws
+
+# IntelliJ
+out/
+
+# mpeltonen/sbt-idea plugin
+.idea_modules/
+
+# JIRA plugin
+atlassian-ide-plugin.xml
+
+# Cursive Clojure plugin
+.idea/replstate.xml
+
+# SonarLint plugin
+.idea/sonarlint/
+
+# Crashlytics plugin (for Android Studio and IntelliJ)
+com_crashlytics_export_strings.xml
+crashlytics.properties
+crashlytics-build.properties
+fabric.properties
+
+# Editor-based Rest Client
+.idea/httpRequests
+
+# Android studio 3.1+ serialized cache file
+.idea/caches/build_file_checksums.ser
+
+### Linux template
+*~
+
+# temporary files which can be created if a process still has a handle open of a deleted file
+.fuse_hidden*
+
+# KDE directory preferences
+.directory
+
+# Linux trash folder which might appear on any partition or disk
+.Trash-*
+
+# .nfs files are created when an open file is removed but is still being accessed
+.nfs*
+
+### Go template
+# If you prefer the allow list template instead of the deny list, see community template:
+# https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore
+#
+# Binaries for programs and plugins
+*.exe
+*.exe~
+*.dll
+*.so
+*.dylib
+
+# Test binary, built with `go test -c`
+*.test
+
+# Output of the go coverage tool, specifically when used with LiteIDE
+*.out
+
+# Dependency directories (remove the comment below to include it)
+# vendor/
+
+# Go workspace file
+go.work
+
+### Windows template
+# Windows thumbnail cache files
+Thumbs.db
+Thumbs.db:encryptable
+ehthumbs.db
+ehthumbs_vista.db
+
+# Dump file
+*.stackdump
+
+# Folder config file
+[Dd]esktop.ini
+
+# Recycle Bin used on file shares
+$RECYCLE.BIN/
+
+# Windows Installer files
+*.cab
+*.msi
+*.msix
+*.msm
+*.msp
+
+# Windows shortcuts
+*.lnk
+
+### macOS template
+# General
+.DS_Store
+.AppleDouble
+.LSOverride
+
+# Icon must end with two \r
+Icon
+
+# Thumbnails
+._*
+
+# Files that might appear in the root of a volume
+.DocumentRevisions-V100
+.fseventsd
+.Spotlight-V100
+.TemporaryItems
+.Trashes
+.VolumeIcon.icns
+.com.apple.timemachine.donotpresent
+
+# Directories potentially created on remote AFP share
+.AppleDB
+.AppleDesktop
+Network Trash Folder
+Temporary Items
+.apdisk
+
+### SublimeText template
+# Cache files for Sublime Text
+*.tmlanguage.cache
+*.tmPreferences.cache
+*.stTheme.cache
+
+# Workspace files are user-specific
+*.sublime-workspace
+
+# Project files should be checked into the repository, unless a significant
+# proportion of contributors will probably not be using Sublime Text
+# *.sublime-project
+
+# SFTP configuration file
+sftp-config.json
+sftp-config-alt*.json
+
+# Package control specific files
+Package Control.last-run
+Package Control.ca-list
+Package Control.ca-bundle
+Package Control.system-ca-bundle
+Package Control.cache/
+Package Control.ca-certs/
+Package Control.merged-ca-bundle
+Package Control.user-ca-bundle
+oscrypto-ca-bundle.crt
+bh_unicode_properties.cache
+
+# Sublime-github package stores a github token in this file
+# https://packagecontrol.io/packages/sublime-github
+GitHub.sublime-settings
+
diff --git a/.idea/.gitignore b/.idea/.gitignore
new file mode 100644
index 0000000..13566b8
--- /dev/null
+++ b/.idea/.gitignore
@@ -0,0 +1,8 @@
+# Default ignored files
+/shelf/
+/workspace.xml
+# Editor-based HTTP Client requests
+/httpRequests/
+# Datasource local storage ignored files
+/dataSources/
+/dataSources.local.xml
diff --git a/.idea/csv-editor.xml b/.idea/csv-editor.xml
new file mode 100644
index 0000000..e85aec4
--- /dev/null
+++ b/.idea/csv-editor.xml
@@ -0,0 +1,23 @@
+
+
+
+
+
+
\ No newline at end of file
diff --git a/.idea/modules.xml b/.idea/modules.xml
new file mode 100644
index 0000000..69fa836
--- /dev/null
+++ b/.idea/modules.xml
@@ -0,0 +1,8 @@
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/.idea/vcs.xml b/.idea/vcs.xml
new file mode 100644
index 0000000..35eb1dd
--- /dev/null
+++ b/.idea/vcs.xml
@@ -0,0 +1,6 @@
+
+
+
+
+
+
\ No newline at end of file
diff --git a/.idea/winworldpc-ipfs-scraper.iml b/.idea/winworldpc-ipfs-scraper.iml
new file mode 100644
index 0000000..5e764c4
--- /dev/null
+++ b/.idea/winworldpc-ipfs-scraper.iml
@@ -0,0 +1,9 @@
+
+
+
+
+
+
+
+
+
\ No newline at end of file
diff --git a/go.mod b/go.mod
new file mode 100644
index 0000000..1ec1688
--- /dev/null
+++ b/go.mod
@@ -0,0 +1,10 @@
+module github.com/Neo-Desktop/winworldpc-ipfs-scraper
+
+go 1.18
+
+require github.com/PuerkitoBio/goquery v1.8.0
+
+require (
+ github.com/andybalholm/cascadia v1.3.1 // indirect
+ golang.org/x/net v0.0.0-20210916014120-12bc252f5db8 // indirect
+)
diff --git a/go.sum b/go.sum
new file mode 100644
index 0000000..d5e55e1
--- /dev/null
+++ b/go.sum
@@ -0,0 +1,11 @@
+github.com/PuerkitoBio/goquery v1.8.0 h1:PJTF7AmFCFKk1N6V6jmKfrNH9tV5pNE6lZMkG0gta/U=
+github.com/PuerkitoBio/goquery v1.8.0/go.mod h1:ypIiRMtY7COPGk+I/YbZLbxsxn9g5ejnI2HSMtkjZvI=
+github.com/andybalholm/cascadia v1.3.1 h1:nhxRkql1kdYCc8Snf7D5/D3spOX+dBgjA6u8x004T2c=
+github.com/andybalholm/cascadia v1.3.1/go.mod h1:R4bJ1UQfqADjvDa4P6HZHLh/3OxWWEqc0Sk8XGwHqvA=
+golang.org/x/net v0.0.0-20210916014120-12bc252f5db8 h1:/6y1LfuqNuQdHAm0jjtPtgRcxIxjVZgm5OTu8/QhZvk=
+golang.org/x/net v0.0.0-20210916014120-12bc252f5db8/go.mod h1:9nx3DQGgdP8bBQD5qxJ1jj9UTztislL4KSBs9R2vV5Y=
+golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
+golang.org/x/sys v0.0.0-20210423082822-04245dca01da/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
+golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
+golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
+golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
diff --git a/main.go b/main.go
new file mode 100644
index 0000000..851149a
--- /dev/null
+++ b/main.go
@@ -0,0 +1,269 @@
+package main
+
+import (
+ "encoding/csv"
+ "fmt"
+ "io"
+ "log"
+ "net/http"
+ "net/url"
+ "os"
+ "strings"
+ "time"
+
+ gq "github.com/PuerkitoBio/goquery"
+)
+
+const (
+ URL = "https://winworldpc.com"
+ csvFileName = "results.csv"
+ csvFullResultsFileName = "results_full.csv"
+ logFileName = "output.log"
+ UserAgent = "Mozilla/5.0 (compatible; IPFS.ScraperBot/v1.0; +https://github.com/Neo-Desktop/winworldpc-ipfs-scraper)"
+)
+
+type File struct {
+ Name string
+ Version string
+ Language string
+ GUID string
+ Size string
+ Hash string
+ Architecture string
+ IPFSLink string
+ MirrorLinks []string
+}
+
+func (f File) MarshalCSV() []string {
+ out := []string{
+ f.Name,
+ f.Version,
+ f.Language,
+ f.GUID,
+ f.Size,
+ f.Hash,
+ f.Architecture,
+ f.IPFSLink,
+ }
+
+ out = append(out, f.MirrorLinks...)
+ return out
+}
+
+type Article struct {
+ Title string
+ Version string
+ WWPCLink string
+ Files []File
+}
+
+func scrapeSearchPage(page int) []Article {
+ log.Printf("=============================== PAGE %2d ===============================", page)
+ urlA, err := url.Parse(URL + "/search")
+ if err != nil {
+ log.Fatal(err)
+ }
+
+ queryParameters := urlA.Query()
+ queryParameters.Add("page", fmt.Sprintf("%d", page))
+ urlA.RawQuery = queryParameters.Encode()
+
+ res, err := fetch(urlA.String())
+ if err != nil {
+ log.Fatal(err)
+ }
+
+ defer res.Body.Close()
+ if res.StatusCode != http.StatusOK {
+ log.Fatalf("status code error: %d %s", res.StatusCode, res.Status)
+ }
+
+ doc, err := gq.NewDocumentFromReader(res.Body)
+ if err != nil {
+ log.Fatal(err)
+ }
+
+ articles := make([]Article, 0)
+
+ doc.Find(".media>.media-body").Each(func(i int, s *gq.Selection) {
+ title := strings.TrimSpace(s.Find(".mt-0 a").First().Text())
+
+ s.Find(".nav>.nav-link>a").Each(func(j int, s1 *gq.Selection) {
+ url, ok := s1.Attr("href")
+ if !ok {
+ log.Printf("anchor does not have a href")
+ return
+ }
+
+ version := strings.TrimSpace(s1.Text())
+
+ if len(articles) > 6 {
+ return
+ }
+ articles = append(articles, scrapeArticlePage(Article{
+ Title: title,
+ Version: version,
+ WWPCLink: url,
+ }))
+ })
+ })
+
+ return articles
+}
+
+func scrapeArticlePage(article Article) Article {
+ res, err := fetch(URL + article.WWPCLink)
+ if err != nil {
+ log.Println(err)
+ return article
+ }
+
+ defer res.Body.Close()
+ if res.StatusCode != http.StatusOK {
+ log.Printf("status code error: %d %s", res.StatusCode, res.Status)
+ return article
+ }
+
+ doc, err := gq.NewDocumentFromReader(res.Body)
+ if err != nil {
+ log.Println(err)
+ return article
+ }
+
+ doc.Find("#downloadsTable tbody tr").Each(func(i int, tr *gq.Selection) {
+ file := File{}
+
+ tr.Find("td").Each(func(j int, td *gq.Selection) {
+ switch j {
+
+ case 0: // download name / link / guid
+ link, ok := td.Find("a").Attr("href")
+ if ok {
+ file.GUID = strings.Replace(link, "/download/", "", -1)
+ }
+ file.Name = strings.TrimSpace(td.Text())
+
+ case 1: // version
+ file.Version = strings.TrimSpace(td.Text())
+
+ case 2: // language
+ file.Language = strings.TrimSpace(td.Text())
+
+ case 3: // architecture
+ title, ok := td.Find("img").Attr("title")
+ if ok {
+ file.Architecture = title
+ }
+
+ case 4: // file size / hash
+ file.Size = strings.TrimSpace(td.Text())
+ hash, ok := td.Find("span").Attr("title")
+ if ok {
+ file.Hash = strings.TrimSpace(hash)
+ }
+
+ //case 5: // download counter
+ }
+ })
+
+ article.Files = append(article.Files, scrapeDownloadPage(file))
+ })
+
+ return article
+}
+
+func scrapeDownloadPage(file File) File {
+ res, err := fetch(URL + "/download/" + file.GUID)
+ if err != nil {
+ log.Println(err)
+ return file
+ }
+
+ defer res.Body.Close()
+ if res.StatusCode != http.StatusOK {
+ log.Printf("status code error: %d %s", res.StatusCode, res.Status)
+ return file
+ }
+
+ doc, err := gq.NewDocumentFromReader(res.Body)
+ if err != nil {
+ log.Println(err)
+ return file
+ }
+
+ link, ok := doc.Find("#localClientLink a").Attr("href")
+ if ok {
+ file.IPFSLink = link
+ }
+
+ doc.Find("#mirrorsList a").Each(func(i int, s *gq.Selection) {
+ link, ok := s.Attr("href")
+ if ok {
+ file.MirrorLinks = append(file.MirrorLinks, link)
+ }
+ })
+
+ return file
+}
+
+func fetch(urlA string) (*http.Response, error) {
+ client := &http.Client{}
+
+ log.Printf("sleeping 3 seconds before requesting %s", urlA)
+ time.Sleep(3 * time.Second)
+
+ req, err := http.NewRequest(http.MethodGet, urlA, nil)
+ if err != nil {
+ return req.Response, err
+ }
+
+ req.Header.Set("User-Agent", UserAgent)
+ return client.Do(req)
+}
+
+func writeCSV(articles []Article, filename string) {
+ csvFile, err := os.OpenFile(filename, os.O_RDWR|os.O_APPEND|os.O_CREATE, 0644)
+ if err != nil {
+ log.Println(err)
+ return
+ }
+
+ defer csvFile.Close()
+
+ csvWriter := csv.NewWriter(csvFile)
+ defer csvWriter.Flush()
+
+ for _, article := range articles {
+ for _, file := range article.Files {
+ _ = csvWriter.Write(file.MarshalCSV())
+ }
+ }
+}
+
+func main() {
+ logHandle, err := os.OpenFile(logFileName, os.O_RDWR|os.O_APPEND|os.O_CREATE, 0644)
+ if err != nil {
+ log.Fatalf("error opening file: %v", err)
+ }
+
+ defer logHandle.Close()
+
+ logWriter := io.MultiWriter(os.Stdout, logHandle)
+ log.SetOutput(logWriter)
+
+ log.Printf("WinWorld IPFS Scraper Started")
+ startTime := time.Now()
+
+ articles := make([]Article, 0)
+
+ for i := 1; i < 63; i++ {
+ results := scrapeSearchPage(i)
+ articles = append(articles, results...)
+ go writeCSV(results, csvFileName)
+ }
+
+ writeCSV(articles, csvFullResultsFileName)
+
+ log.Printf("WinWorld IPFS Scraper completed")
+ log.Printf("Total time: %s", time.Now().Sub(startTime).String())
+}