initial commit, working code

This commit is contained in:
Neo committed 2023-01-28 10:27:43 -08:00
1 parent 2a469c48e6
commit fc36b0eaae
9 files changed
+569

No files matched your search

+225
View File
@@ -0,0 +1,225 @@
*.csv
*.log
### NotepadPP template
# Notepad++ backups #
*.bak
### VisualStudioCode template
.vscode/*
!.vscode/settings.json
!.vscode/tasks.json
!.vscode/launch.json
!.vscode/extensions.json
!.vscode/*.code-snippets
# Local History for Visual Studio Code
.history/
# Built Visual Studio Code Extensions
*.vsix
### JetBrains template
# Covers JetBrains IDEs: IntelliJ, RubyMine, PhpStorm, AppCode, PyCharm, CLion, Android Studio, WebStorm and Rider
# Reference: https://intellij-support.jetbrains.com/hc/en-us/articles/206544839
# User-specific stuff
.idea/**/workspace.xml
.idea/**/tasks.xml
.idea/**/usage.statistics.xml
.idea/**/dictionaries
.idea/**/shelf
# AWS User-specific
.idea/**/aws.xml
# Generated files
.idea/**/contentModel.xml
# Sensitive or high-churn files
.idea/**/dataSources/
.idea/**/dataSources.ids
.idea/**/dataSources.local.xml
.idea/**/sqlDataSources.xml
.idea/**/dynamic.xml
.idea/**/uiDesigner.xml
.idea/**/dbnavigator.xml
# Gradle
.idea/**/gradle.xml
.idea/**/libraries
# Gradle and Maven with auto-import
# When using Gradle or Maven with auto-import, you should exclude module files,
# since they will be recreated, and may cause churn. Uncomment if using
# auto-import.
# .idea/artifacts
# .idea/compiler.xml
# .idea/jarRepositories.xml
# .idea/modules.xml
# .idea/*.iml
# .idea/modules
# *.iml
# *.ipr
# CMake
cmake-build-*/
# Mongo Explorer plugin
.idea/**/mongoSettings.xml
# File-based project format
*.iws
# IntelliJ
out/
# mpeltonen/sbt-idea plugin
.idea_modules/
# JIRA plugin
atlassian-ide-plugin.xml
# Cursive Clojure plugin
.idea/replstate.xml
# SonarLint plugin
.idea/sonarlint/
# Crashlytics plugin (for Android Studio and IntelliJ)
com_crashlytics_export_strings.xml
crashlytics.properties
crashlytics-build.properties
fabric.properties
# Editor-based Rest Client
.idea/httpRequests
# Android studio 3.1+ serialized cache file
.idea/caches/build_file_checksums.ser
### Linux template
*~
# temporary files which can be created if a process still has a handle open of a deleted file
.fuse_hidden*
# KDE directory preferences
.directory
# Linux trash folder which might appear on any partition or disk
.Trash-*
# .nfs files are created when an open file is removed but is still being accessed
.nfs*
### Go template
# If you prefer the allow list template instead of the deny list, see community template:
# https://github.com/github/gitignore/blob/main/community/Golang/Go.AllowList.gitignore
#
# Binaries for programs and plugins
*.exe
*.exe~
*.dll
*.so
*.dylib
# Test binary, built with `go test -c`
*.test
# Output of the go coverage tool, specifically when used with LiteIDE
*.out
# Dependency directories (remove the comment below to include it)
# vendor/
# Go workspace file
go.work
### Windows template
# Windows thumbnail cache files
Thumbs.db
Thumbs.db:encryptable
ehthumbs.db
ehthumbs_vista.db
# Dump file
*.stackdump
# Folder config file
[Dd]esktop.ini
# Recycle Bin used on file shares
$RECYCLE.BIN/
# Windows Installer files
*.cab
*.msi
*.msix
*.msm
*.msp
# Windows shortcuts
*.lnk
### macOS template
# General
.DS_Store
.AppleDouble
.LSOverride
# Icon must end with two \r
Icon
# Thumbnails
._*
# Files that might appear in the root of a volume
.DocumentRevisions-V100
.fseventsd
.Spotlight-V100
.TemporaryItems
.Trashes
.VolumeIcon.icns
.com.apple.timemachine.donotpresent
# Directories potentially created on remote AFP share
.AppleDB
.AppleDesktop
Network Trash Folder
Temporary Items
.apdisk
### SublimeText template
# Cache files for Sublime Text
*.tmlanguage.cache
*.tmPreferences.cache
*.stTheme.cache
# Workspace files are user-specific
*.sublime-workspace
# Project files should be checked into the repository, unless a significant
# proportion of contributors will probably not be using Sublime Text
# *.sublime-project
# SFTP configuration file
sftp-config.json
sftp-config-alt*.json
# Package control specific files
Package Control.last-run
Package Control.ca-list
Package Control.ca-bundle
Package Control.system-ca-bundle
Package Control.cache/
Package Control.ca-certs/
Package Control.merged-ca-bundle
Package Control.user-ca-bundle
oscrypto-ca-bundle.crt
bh_unicode_properties.cache
# Sublime-github package stores a github token in this file
# https://packagecontrol.io/packages/sublime-github
GitHub.sublime-settings
+8
View File
@@ -0,0 +1,8 @@
# Default ignored files
/shelf/
/workspace.xml
# Editor-based HTTP Client requests
/httpRequests/
# Datasource local storage ignored files
/dataSources/
/dataSources.local.xml
+23
View File
@@ -0,0 +1,23 @@
<?xml version="1.0" encoding="UTF-8"?>
<project version="4">
<component name="CsvFileAttributes">
<option name="attributeMap">
<map>
<entry key="\results.csv">
<value>
<Attribute>
<option name="separator" value="," />
</Attribute>
</value>
</entry>
<entry key="\results_full.csv">
<value>
<Attribute>
<option name="separator" value="," />
</Attribute>
</value>
</entry>
</map>
</option>
</component>
</project>
+8
View File
@@ -0,0 +1,8 @@
<?xml version="1.0" encoding="UTF-8"?>
<project version="4">
<component name="ProjectModuleManager">
<modules>
<module fileurl="file://$PROJECT_DIR$/.idea/winworldpc-ipfs-scraper.iml" filepath="$PROJECT_DIR$/.idea/winworldpc-ipfs-scraper.iml" />
</modules>
</component>
</project>
Generated
+6
View File
@@ -0,0 +1,6 @@
<?xml version="1.0" encoding="UTF-8"?>
<project version="4">
<component name="VcsDirectoryMappings">
<mapping directory="" vcs="Git" />
</component>
</project>
+9
View File
@@ -0,0 +1,9 @@
<?xml version="1.0" encoding="UTF-8"?>
<module type="WEB_MODULE" version="4">
<component name="Go" enabled="true" />
<component name="NewModuleRootManager">
<content url="file://$MODULE_DIR$" />
<orderEntry type="inheritedJdk" />
<orderEntry type="sourceFolder" forTests="false" />
</component>
</module>
+10
View File
@@ -0,0 +1,10 @@
module github.com/Neo-Desktop/winworldpc-ipfs-scraper
go 1.18
require github.com/PuerkitoBio/goquery v1.8.0
require (
github.com/andybalholm/cascadia v1.3.1 // indirect
golang.org/x/net v0.0.0-20210916014120-12bc252f5db8 // indirect
)
+11
View File
@@ -0,0 +1,11 @@
github.com/PuerkitoBio/goquery v1.8.0 h1:PJTF7AmFCFKk1N6V6jmKfrNH9tV5pNE6lZMkG0gta/U=
github.com/PuerkitoBio/goquery v1.8.0/go.mod h1:ypIiRMtY7COPGk+I/YbZLbxsxn9g5ejnI2HSMtkjZvI=
github.com/andybalholm/cascadia v1.3.1 h1:nhxRkql1kdYCc8Snf7D5/D3spOX+dBgjA6u8x004T2c=
github.com/andybalholm/cascadia v1.3.1/go.mod h1:R4bJ1UQfqADjvDa4P6HZHLh/3OxWWEqc0Sk8XGwHqvA=
golang.org/x/net v0.0.0-20210916014120-12bc252f5db8 h1:/6y1LfuqNuQdHAm0jjtPtgRcxIxjVZgm5OTu8/QhZvk=
golang.org/x/net v0.0.0-20210916014120-12bc252f5db8/go.mod h1:9nx3DQGgdP8bBQD5qxJ1jj9UTztislL4KSBs9R2vV5Y=
golang.org/x/sys v0.0.0-20201119102817-f84b799fce68/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/sys v0.0.0-20210423082822-04245dca01da/go.mod h1:h1NjWce9XRLGQEsW7wpKNCjG9DtNlClVuFLEZdDNbEs=
golang.org/x/term v0.0.0-20201126162022-7de9c90e9dd1/go.mod h1:bj7SfCRtBDWHUb9snDiAeCFNEtKQo2Wmx5Cou7ajbmo=
golang.org/x/text v0.3.6/go.mod h1:5Zoc/QRtKVWzQhOtBMvqHzDpF6irO9z98xDceosuGiQ=
golang.org/x/tools v0.0.0-20180917221912-90fa682c2a6e/go.mod h1:n7NCudcB/nEzxVGmLbDWY5pfWTLqBcC2KZ6jyYvM4mQ=
+269
View File
@@ -0,0 +1,269 @@
package main
import (
"encoding/csv"
"fmt"
"io"
"log"
"net/http"
"net/url"
"os"
"strings"
"time"
gq "github.com/PuerkitoBio/goquery"
)
const (
URL = "https://winworldpc.com"
csvFileName = "results.csv"
csvFullResultsFileName = "results_full.csv"
logFileName = "output.log"
UserAgent = "Mozilla/5.0 (compatible; IPFS.ScraperBot/v1.0; +https://github.com/Neo-Desktop/winworldpc-ipfs-scraper)"
)
type File struct {
Name string
Version string
Language string
GUID string
Size string
Hash string
Architecture string
IPFSLink string
MirrorLinks []string
}
func (f File) MarshalCSV() []string {
out := []string{
f.Name,
f.Version,
f.Language,
f.GUID,
f.Size,
f.Hash,
f.Architecture,
f.IPFSLink,
}
out = append(out, f.MirrorLinks...)
return out
}
type Article struct {
Title string
Version string
WWPCLink string
Files []File
}
func scrapeSearchPage(page int) []Article {
log.Printf("=============================== PAGE %2d ===============================", page)
urlA, err := url.Parse(URL + "/search")
if err != nil {
log.Fatal(err)
}
queryParameters := urlA.Query()
queryParameters.Add("page", fmt.Sprintf("%d", page))
urlA.RawQuery = queryParameters.Encode()
res, err := fetch(urlA.String())
if err != nil {
log.Fatal(err)
}
defer res.Body.Close()
if res.StatusCode != http.StatusOK {
log.Fatalf("status code error: %d %s", res.StatusCode, res.Status)
}
doc, err := gq.NewDocumentFromReader(res.Body)
if err != nil {
log.Fatal(err)
}
articles := make([]Article, 0)
doc.Find(".media>.media-body").Each(func(i int, s *gq.Selection) {
title := strings.TrimSpace(s.Find(".mt-0 a").First().Text())
s.Find(".nav>.nav-link>a").Each(func(j int, s1 *gq.Selection) {
url, ok := s1.Attr("href")
if !ok {
log.Printf("anchor does not have a href")
return
}
version := strings.TrimSpace(s1.Text())
if len(articles) > 6 {
return
}
articles = append(articles, scrapeArticlePage(Article{
Title: title,
Version: version,
WWPCLink: url,
}))
})
})
return articles
}
func scrapeArticlePage(article Article) Article {
res, err := fetch(URL + article.WWPCLink)
if err != nil {
log.Println(err)
return article
}
defer res.Body.Close()
if res.StatusCode != http.StatusOK {
log.Printf("status code error: %d %s", res.StatusCode, res.Status)
return article
}
doc, err := gq.NewDocumentFromReader(res.Body)
if err != nil {
log.Println(err)
return article
}
doc.Find("#downloadsTable tbody tr").Each(func(i int, tr *gq.Selection) {
file := File{}
tr.Find("td").Each(func(j int, td *gq.Selection) {
switch j {
case 0: // download name / link / guid
link, ok := td.Find("a").Attr("href")
if ok {
file.GUID = strings.Replace(link, "/download/", "", -1)
}
file.Name = strings.TrimSpace(td.Text())
case 1: // version
file.Version = strings.TrimSpace(td.Text())
case 2: // language
file.Language = strings.TrimSpace(td.Text())
case 3: // architecture
title, ok := td.Find("img").Attr("title")
if ok {
file.Architecture = title
}
case 4: // file size / hash
file.Size = strings.TrimSpace(td.Text())
hash, ok := td.Find("span").Attr("title")
if ok {
file.Hash = strings.TrimSpace(hash)
}
//case 5: // download counter
}
})
article.Files = append(article.Files, scrapeDownloadPage(file))
})
return article
}
func scrapeDownloadPage(file File) File {
res, err := fetch(URL + "/download/" + file.GUID)
if err != nil {
log.Println(err)
return file
}
defer res.Body.Close()
if res.StatusCode != http.StatusOK {
log.Printf("status code error: %d %s", res.StatusCode, res.Status)
return file
}
doc, err := gq.NewDocumentFromReader(res.Body)
if err != nil {
log.Println(err)
return file
}
link, ok := doc.Find("#localClientLink a").Attr("href")
if ok {
file.IPFSLink = link
}
doc.Find("#mirrorsList a").Each(func(i int, s *gq.Selection) {
link, ok := s.Attr("href")
if ok {
file.MirrorLinks = append(file.MirrorLinks, link)
}
})
return file
}
func fetch(urlA string) (*http.Response, error) {
client := &http.Client{}
log.Printf("sleeping 3 seconds before requesting %s", urlA)
time.Sleep(3 * time.Second)
req, err := http.NewRequest(http.MethodGet, urlA, nil)
if err != nil {
return req.Response, err
}
req.Header.Set("User-Agent", UserAgent)
return client.Do(req)
}
func writeCSV(articles []Article, filename string) {
csvFile, err := os.OpenFile(filename, os.O_RDWR|os.O_APPEND|os.O_CREATE, 0644)
if err != nil {
log.Println(err)
return
}
defer csvFile.Close()
csvWriter := csv.NewWriter(csvFile)
defer csvWriter.Flush()
for _, article := range articles {
for _, file := range article.Files {
_ = csvWriter.Write(file.MarshalCSV())
}
}
}
func main() {
logHandle, err := os.OpenFile(logFileName, os.O_RDWR|os.O_APPEND|os.O_CREATE, 0644)
if err != nil {
log.Fatalf("error opening file: %v", err)
}
defer logHandle.Close()
logWriter := io.MultiWriter(os.Stdout, logHandle)
log.SetOutput(logWriter)
log.Printf("WinWorld IPFS Scraper Started")
startTime := time.Now()
articles := make([]Article, 0)
for i := 1; i < 63; i++ {
results := scrapeSearchPage(i)
articles = append(articles, results...)
go writeCSV(results, csvFileName)
}
writeCSV(articles, csvFullResultsFileName)
log.Printf("WinWorld IPFS Scraper completed")
log.Printf("Total time: %s", time.Now().Sub(startTime).String())
}