diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..dd3b96b --- /dev/null +++ b/.gitignore @@ -0,0 +1,225 @@ +*.csv +*.log + +### NotepadPP template +# Notepad++ backups # +*.bak + +### VisualStudioCode template +.vscode/* +!.vscode/settings.json +!.vscode/tasks.json +!.vscode/launch.json +!.vscode/extensions.json +!.vscode/*.code-snippets + +# Local History for Visual Studio Code +.history/ + +# Built Visual Studio Code Extensions +*.vsix + +### JetBrains template +# Covers JetBrains IDEs: IntelliJ, RubyMine, PhpStorm, AppCode, PyCharm, CLion, Android Studio, WebStorm and Rider +# Reference: https://intellij-support.jetbrains.com/hc/en-us/articles/206544839 + +# User-specific stuff +.idea/**/workspace.xml +.idea/**/tasks.xml +.idea/**/usage.statistics.xml +.idea/**/dictionaries +.idea/**/shelf + +# AWS User-specific +.idea/**/aws.xml + +# Generated files +.idea/**/contentModel.xml + +# Sensitive or high-churn files +.idea/**/dataSources/ +.idea/**/dataSources.ids +.idea/**/dataSources.local.xml +.idea/**/sqlDataSources.xml +.idea/**/dynamic.xml +.idea/**/uiDesigner.xml +.idea/**/dbnavigator.xml + +# Gradle +.idea/**/gradle.xml +.idea/**/libraries + +# Gradle and Maven with auto-import +# When using Gradle or Maven with auto-import, you should exclude module files, +# since they will be recreated, and may cause churn. Uncomment if using +# auto-import. +# .idea/artifacts +# .idea/compiler.xml +# .idea/jarRepositories.xml +# .idea/modules.xml +# .idea/*.iml +# .idea/modules +# *.iml +# *.ipr + +# CMake +cmake-build-*/ + +# Mongo Explorer plugin +.idea/**/mongoSettings.xml + +# File-based project format +*.iws + +# IntelliJ +out/ + +# mpeltonen/sbt-idea plugin +.idea_modules/ + +# JIRA plugin +atlassian-ide-plugin.xml + +# Cursive Clojure plugin +.idea/replstate.xml + +# SonarLint plugin +.idea/sonarlint/ + +# Crashlytics plugin (for Android Studio and IntelliJ) +com_crashlytics_export_strings.xml +crashlytics.properties +crashlytics-build.properties +fabric.properties + +# Editor-based Rest Client +.idea/httpRequests + +# Android studio 3.1+ serialized cache file +.idea/caches/build_file_checksums.ser + +### Linux template +*~ + +# temporary files which can be created if a process still has a handle open of a deleted file +.fuse_hidden* + +# KDE directory preferences +.directory + +# Linux trash folder which might appear on any partition or disk +.Trash-* + +# .nfs files are created when an open file is removed but is still being accessed +.nfs* + +### Go template +# If you prefer the allow list template instead of the deny list, see community template: +# https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore +# +# Binaries for programs and plugins +*.exe +*.exe~ +*.dll +*.so +*.dylib + +# Test binary, built with `go test -c` +*.test + +# Output of the go coverage tool, specifically when used with LiteIDE +*.out + +# Dependency directories (remove the comment below to include it) +# vendor/ + +# Go workspace file +go.work + +### Windows template +# Windows thumbnail cache files +Thumbs.db +Thumbs.db:encryptable +ehthumbs.db +ehthumbs_vista.db + +# Dump file +*.stackdump + +# Folder config file +[Dd]esktop.ini + +# Recycle Bin used on file shares +$RECYCLE.BIN/ + +# Windows Installer files +*.cab +*.msi +*.msix +*.msm +*.msp + +# Windows shortcuts +*.lnk + +### macOS template +# General +.DS_Store +.AppleDouble +.LSOverride + +# Icon must end with two \r +Icon + +# Thumbnails +._* + +# Files that might appear in the root of a volume +.DocumentRevisions-V100 +.fseventsd +.Spotlight-V100 +.TemporaryItems +.Trashes +.VolumeIcon.icns +.com.apple.timemachine.donotpresent + +# Directories potentially created on remote AFP share +.AppleDB +.AppleDesktop +Network Trash Folder +Temporary Items +.apdisk + +### SublimeText template +# Cache files for Sublime Text +*.tmlanguage.cache +*.tmPreferences.cache +*.stTheme.cache + +# Workspace files are user-specific +*.sublime-workspace + +# Project files should be checked into the repository, unless a significant +# proportion of contributors will probably not be using Sublime Text +# *.sublime-project + +# SFTP configuration file +sftp-config.json +sftp-config-alt*.json + +# Package control specific files +Package Control.last-run +Package Control.ca-list +Package Control.ca-bundle +Package Control.system-ca-bundle +Package Control.cache/ +Package Control.ca-certs/ +Package Control.merged-ca-bundle +Package Control.user-ca-bundle +oscrypto-ca-bundle.crt +bh_unicode_properties.cache + +# Sublime-github package stores a github token in this file +# https://packagecontrol.io/packages/sublime-github +GitHub.sublime-settings + diff --git a/.idea/.gitignore b/.idea/.gitignore new file mode 100644 index 0000000..13566b8 --- /dev/null +++ b/.idea/.gitignore @@ -0,0 +1,8 @@ +# Default ignored files +/shelf/ +/workspace.xml +# Editor-based HTTP Client requests +/httpRequests/ +# Datasource local storage ignored files +/dataSources/ +/dataSources.local.xml diff --git a/.idea/csv-editor.xml b/.idea/csv-editor.xml new file mode 100644 index 0000000..e85aec4 --- /dev/null +++ b/.idea/csv-editor.xml @@ -0,0 +1,23 @@ + + + + + + \ No newline at end of file diff --git a/.idea/modules.xml b/.idea/modules.xml new file mode 100644 index 0000000..69fa836 --- /dev/null +++ b/.idea/modules.xml @@ -0,0 +1,8 @@ + + + + + + + + \ No newline at end of file diff --git a/.idea/vcs.xml b/.idea/vcs.xml new file mode 100644 index 0000000..35eb1dd --- /dev/null +++ b/.idea/vcs.xml @@ -0,0 +1,6 @@ + + + + + + \ No newline at end of file diff --git a/.idea/winworldpc-ipfs-scraper.iml b/.idea/winworldpc-ipfs-scraper.iml new file mode 100644 index 0000000..5e764c4 --- /dev/null +++ b/.idea/winworldpc-ipfs-scraper.iml @@ -0,0 +1,9 @@ + + + + + + + + + \ No newline at end of file diff --git a/go.mod b/go.mod new file mode 100644 index 0000000..1ec1688 --- /dev/null +++ b/go.mod @@ -0,0 +1,10 @@ +module github.com/Neo-Desktop/winworldpc-ipfs-scraper + +go 1.18 + +require github.com/PuerkitoBio/goquery v1.8.0 + +require ( + github.com/andybalholm/cascadia v1.3.1 // indirect + golang.org/x/net v0.0.0-20210916014120-12bc252f5db8 // indirect +) diff --git a/go.sum b/go.sum new file mode 100644 index 0000000..d5e55e1 --- /dev/null +++ b/go.sum @@ -0,0 +1,11 @@ +github.com/PuerkitoBio/goquery v1.8.0 h1:PJTF7AmFCFKk1N6V6jmKfrNH9tV5pNE6lZMkG0gta/U= +github.com/PuerkitoBio/goquery v1.8.0/go.mod h1:ypIiRMtY7COPGk+I/YbZLbxsxn9g5ejnI2HSMtkjZvI= +github.com/andybalholm/cascadia v1.3.1 h1:nhxRkql1kdYCc8Snf7D5/D3spOX+dBgjA6u8x004T2c= +github.com/andybalholm/cascadia v1.3.1/go.mod h1:R4bJ1UQfqADjvDa4P6HZHLh/3OxWWEqc0Sk8XGwHqvA= +golang.org/x/net v0.0.0-20210916014120-12bc252f5db8 h1:/6y1LfuqNuQdHAm0jjtPtgRcxIxjVZgm5OTu8/QhZvk= +golang.org/x/net v0.0.0-20210916014120-12bc252f5db8/go.mod h1:9nx3DQGgdP8bBQD5qxJ1jj9UTztislL4KSBs9R2vV5Y= +golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/sys v0.0.0-20210423082822-04245dca01da/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs= +golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo= +golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ= +golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ= diff --git a/main.go b/main.go new file mode 100644 index 0000000..851149a --- /dev/null +++ b/main.go @@ -0,0 +1,269 @@ +package main + +import ( + "encoding/csv" + "fmt" + "io" + "log" + "net/http" + "net/url" + "os" + "strings" + "time" + + gq "github.com/PuerkitoBio/goquery" +) + +const ( + URL = "https://winworldpc.com" + csvFileName = "results.csv" + csvFullResultsFileName = "results_full.csv" + logFileName = "output.log" + UserAgent = "Mozilla/5.0 (compatible; IPFS.ScraperBot/v1.0; +https://github.com/Neo-Desktop/winworldpc-ipfs-scraper)" +) + +type File struct { + Name string + Version string + Language string + GUID string + Size string + Hash string + Architecture string + IPFSLink string + MirrorLinks []string +} + +func (f File) MarshalCSV() []string { + out := []string{ + f.Name, + f.Version, + f.Language, + f.GUID, + f.Size, + f.Hash, + f.Architecture, + f.IPFSLink, + } + + out = append(out, f.MirrorLinks...) + return out +} + +type Article struct { + Title string + Version string + WWPCLink string + Files []File +} + +func scrapeSearchPage(page int) []Article { + log.Printf("=============================== PAGE %2d ===============================", page) + urlA, err := url.Parse(URL + "/search") + if err != nil { + log.Fatal(err) + } + + queryParameters := urlA.Query() + queryParameters.Add("page", fmt.Sprintf("%d", page)) + urlA.RawQuery = queryParameters.Encode() + + res, err := fetch(urlA.String()) + if err != nil { + log.Fatal(err) + } + + defer res.Body.Close() + if res.StatusCode != http.StatusOK { + log.Fatalf("status code error: %d %s", res.StatusCode, res.Status) + } + + doc, err := gq.NewDocumentFromReader(res.Body) + if err != nil { + log.Fatal(err) + } + + articles := make([]Article, 0) + + doc.Find(".media>.media-body").Each(func(i int, s *gq.Selection) { + title := strings.TrimSpace(s.Find(".mt-0 a").First().Text()) + + s.Find(".nav>.nav-link>a").Each(func(j int, s1 *gq.Selection) { + url, ok := s1.Attr("href") + if !ok { + log.Printf("anchor does not have a href") + return + } + + version := strings.TrimSpace(s1.Text()) + + if len(articles) > 6 { + return + } + articles = append(articles, scrapeArticlePage(Article{ + Title: title, + Version: version, + WWPCLink: url, + })) + }) + }) + + return articles +} + +func scrapeArticlePage(article Article) Article { + res, err := fetch(URL + article.WWPCLink) + if err != nil { + log.Println(err) + return article + } + + defer res.Body.Close() + if res.StatusCode != http.StatusOK { + log.Printf("status code error: %d %s", res.StatusCode, res.Status) + return article + } + + doc, err := gq.NewDocumentFromReader(res.Body) + if err != nil { + log.Println(err) + return article + } + + doc.Find("#downloadsTable tbody tr").Each(func(i int, tr *gq.Selection) { + file := File{} + + tr.Find("td").Each(func(j int, td *gq.Selection) { + switch j { + + case 0: // download name / link / guid + link, ok := td.Find("a").Attr("href") + if ok { + file.GUID = strings.Replace(link, "/download/", "", -1) + } + file.Name = strings.TrimSpace(td.Text()) + + case 1: // version + file.Version = strings.TrimSpace(td.Text()) + + case 2: // language + file.Language = strings.TrimSpace(td.Text()) + + case 3: // architecture + title, ok := td.Find("img").Attr("title") + if ok { + file.Architecture = title + } + + case 4: // file size / hash + file.Size = strings.TrimSpace(td.Text()) + hash, ok := td.Find("span").Attr("title") + if ok { + file.Hash = strings.TrimSpace(hash) + } + + //case 5: // download counter + } + }) + + article.Files = append(article.Files, scrapeDownloadPage(file)) + }) + + return article +} + +func scrapeDownloadPage(file File) File { + res, err := fetch(URL + "/download/" + file.GUID) + if err != nil { + log.Println(err) + return file + } + + defer res.Body.Close() + if res.StatusCode != http.StatusOK { + log.Printf("status code error: %d %s", res.StatusCode, res.Status) + return file + } + + doc, err := gq.NewDocumentFromReader(res.Body) + if err != nil { + log.Println(err) + return file + } + + link, ok := doc.Find("#localClientLink a").Attr("href") + if ok { + file.IPFSLink = link + } + + doc.Find("#mirrorsList a").Each(func(i int, s *gq.Selection) { + link, ok := s.Attr("href") + if ok { + file.MirrorLinks = append(file.MirrorLinks, link) + } + }) + + return file +} + +func fetch(urlA string) (*http.Response, error) { + client := &http.Client{} + + log.Printf("sleeping 3 seconds before requesting %s", urlA) + time.Sleep(3 * time.Second) + + req, err := http.NewRequest(http.MethodGet, urlA, nil) + if err != nil { + return req.Response, err + } + + req.Header.Set("User-Agent", UserAgent) + return client.Do(req) +} + +func writeCSV(articles []Article, filename string) { + csvFile, err := os.OpenFile(filename, os.O_RDWR|os.O_APPEND|os.O_CREATE, 0644) + if err != nil { + log.Println(err) + return + } + + defer csvFile.Close() + + csvWriter := csv.NewWriter(csvFile) + defer csvWriter.Flush() + + for _, article := range articles { + for _, file := range article.Files { + _ = csvWriter.Write(file.MarshalCSV()) + } + } +} + +func main() { + logHandle, err := os.OpenFile(logFileName, os.O_RDWR|os.O_APPEND|os.O_CREATE, 0644) + if err != nil { + log.Fatalf("error opening file: %v", err) + } + + defer logHandle.Close() + + logWriter := io.MultiWriter(os.Stdout, logHandle) + log.SetOutput(logWriter) + + log.Printf("WinWorld IPFS Scraper Started") + startTime := time.Now() + + articles := make([]Article, 0) + + for i := 1; i < 63; i++ { + results := scrapeSearchPage(i) + articles = append(articles, results...) + go writeCSV(results, csvFileName) + } + + writeCSV(articles, csvFullResultsFileName) + + log.Printf("WinWorld IPFS Scraper completed") + log.Printf("Total time: %s", time.Now().Sub(startTime).String()) +}