From b87ac1bb0e8b03e75103e07e3df3764b49bd2f0a Mon Sep 17 00:00:00 2001 From: Amrit Panesar Date: Thu, 29 Oct 2020 15:17:33 -0700 Subject: [PATCH] add functionality --- .gitignore | 560 ++++++++++++++++++++++++++++++++++++++++++++++++++ message.go | 16 ++ page.go | 250 ++++++++++++++++++++++ scraper.go | 159 ++++++++++++++ serverInfo.go | 7 + 5 files changed, 992 insertions(+) diff --git a/.gitignore b/.gitignore index bfa6a22..17c9c03 100644 --- a/.gitignore +++ b/.gitignore @@ -1 +1,561 @@ # Created by .ignore support plugin (hsz.mobi) +### Go template +# Binaries for programs and plugins +*.exe +*.exe~ +*.dll +*.so +*.dylib + +# Test binary, built with `go test -c` +*.test + +# Output of the go coverage tool, specifically when used with LiteIDE +*.out + +# Dependency directories (remove the comment below to include it) +# vendor/ + +### Linux template +*~ + +# temporary files which can be created if a process still has a handle open of a deleted file +.fuse_hidden* + +# KDE directory preferences +.directory + +# Linux trash folder which might appear on any partition or disk +.Trash-* + +# .nfs files are created when an open file is removed but is still being accessed +.nfs* + +### macOS template +# General +.DS_Store +.AppleDouble +.LSOverride + +# Icon must end with two \r +Icon + +# Thumbnails +._* + +# Files that might appear in the root of a volume +.DocumentRevisions-V100 +.fseventsd +.Spotlight-V100 +.TemporaryItems +.Trashes +.VolumeIcon.icns +.com.apple.timemachine.donotpresent + +# Directories potentially created on remote AFP share +.AppleDB +.AppleDesktop +Network Trash Folder +Temporary Items +.apdisk + +### SublimeText template +# Cache files for Sublime Text +*.tmlanguage.cache +*.tmPreferences.cache +*.stTheme.cache + +# Workspace files are user-specific +*.sublime-workspace + +# Project files should be checked into the repository, unless a significant +# proportion of contributors will probably not be using Sublime Text +# *.sublime-project + +# SFTP configuration file +sftp-config.json +sftp-config-alt*.json + +# Package control specific files +Package Control.last-run +Package Control.ca-list +Package Control.ca-bundle +Package Control.system-ca-bundle +Package Control.cache/ +Package Control.ca-certs/ +Package Control.merged-ca-bundle +Package Control.user-ca-bundle +oscrypto-ca-bundle.crt +bh_unicode_properties.cache + +# Sublime-github package stores a github token in this file +# https://packagecontrol.io/packages/sublime-github +GitHub.sublime-settings + +### JetBrains template +# Covers JetBrains IDEs: IntelliJ, RubyMine, PhpStorm, AppCode, PyCharm, CLion, Android Studio, WebStorm and Rider +# Reference: https://intellij-support.jetbrains.com/hc/en-us/articles/206544839 + +# User-specific stuff +.idea/**/workspace.xml +.idea/**/tasks.xml +.idea/**/usage.statistics.xml +.idea/**/dictionaries +.idea/**/shelf + +# Generated files +.idea/**/contentModel.xml + +# Sensitive or high-churn files +.idea/**/dataSources/ +.idea/**/dataSources.ids +.idea/**/dataSources.local.xml +.idea/**/sqlDataSources.xml +.idea/**/dynamic.xml +.idea/**/uiDesigner.xml +.idea/**/dbnavigator.xml + +# Gradle +.idea/**/gradle.xml +.idea/**/libraries + +# Gradle and Maven with auto-import +# When using Gradle or Maven with auto-import, you should exclude module files, +# since they will be recreated, and may cause churn. Uncomment if using +# auto-import. +# .idea/artifacts +# .idea/compiler.xml +# .idea/jarRepositories.xml +# .idea/modules.xml +# .idea/*.iml +# .idea/modules +# *.iml +# *.ipr + +# CMake +cmake-build-*/ + +# Mongo Explorer plugin +.idea/**/mongoSettings.xml + +# File-based project format +*.iws + +# IntelliJ +out/ + +# mpeltonen/sbt-idea plugin +.idea_modules/ + +# JIRA plugin +atlassian-ide-plugin.xml + +# Cursive Clojure plugin +.idea/replstate.xml + +# Crashlytics plugin (for Android Studio and IntelliJ) +com_crashlytics_export_strings.xml +crashlytics.properties +crashlytics-build.properties +fabric.properties + +# Editor-based Rest Client +.idea/httpRequests + +# Android studio 3.1+ serialized cache file +.idea/caches/build_file_checksums.ser + +### NotepadPP template +# Notepad++ backups # +*.bak + +### VisualStudio template +## Ignore Visual Studio temporary files, build results, and +## files generated by popular Visual Studio add-ons. +## +## Get latest from https://github.com/github/gitignore/blob/master/VisualStudio.gitignore + +# User-specific files +*.rsuser +*.suo +*.user +*.userosscache +*.sln.docstates + +# User-specific files (MonoDevelop/Xamarin Studio) +*.userprefs + +# Mono auto generated files +mono_crash.* + +# Build results +[Dd]ebug/ +[Dd]ebugPublic/ +[Rr]elease/ +[Rr]eleases/ +x64/ +x86/ +[Ww][Ii][Nn]32/ +[Aa][Rr][Mm]/ +[Aa][Rr][Mm]64/ +bld/ +[Bb]in/ +[Oo]bj/ +[Ll]og/ +[Ll]ogs/ + +# Visual Studio 2015/2017 cache/options directory +.vs/ +# Uncomment if you have tasks that create the project's static files in wwwroot +#wwwroot/ + +# Visual Studio 2017 auto generated files +Generated\ Files/ + +# MSTest test Results +[Tt]est[Rr]esult*/ +[Bb]uild[Ll]og.* + +# NUnit +*.VisualState.xml +TestResult.xml +nunit-*.xml + +# Build Results of an ATL Project +[Dd]ebugPS/ +[Rr]eleasePS/ +dlldata.c + +# Benchmark Results +BenchmarkDotNet.Artifacts/ + +# .NET Core +project.lock.json +project.fragment.lock.json +artifacts/ + +# ASP.NET Scaffolding +ScaffoldingReadMe.txt + +# StyleCop +StyleCopReport.xml + +# Files built by Visual Studio +*_i.c +*_p.c +*_h.h +*.ilk +*.meta +*.obj +*.iobj +*.pch +*.pdb +*.ipdb +*.pgc +*.pgd +*.rsp +*.sbr +*.tlb +*.tli +*.tlh +*.tmp +*.tmp_proj +*_wpftmp.csproj +*.log +*.vspscc +*.vssscc +.builds +*.pidb +*.svclog +*.scc + +# Chutzpah Test files +_Chutzpah* + +# Visual C++ cache files +ipch/ +*.aps +*.ncb +*.opendb +*.opensdf +*.sdf +*.cachefile +*.VC.db +*.VC.VC.opendb + +# Visual Studio profiler +*.psess +*.vsp +*.vspx +*.sap + +# Visual Studio Trace Files +*.e2e + +# TFS 2012 Local Workspace +$tf/ + +# Guidance Automation Toolkit +*.gpState + +# ReSharper is a .NET coding add-in +_ReSharper*/ +*.[Rr]e[Ss]harper +*.DotSettings.user + +# TeamCity is a build add-in +_TeamCity* + +# DotCover is a Code Coverage Tool +*.dotCover + +# AxoCover is a Code Coverage Tool +.axoCover/* +!.axoCover/settings.json + +# Coverlet is a free, cross platform Code Coverage Tool +coverage*.json +coverage*.xml +coverage*.info + +# Visual Studio code coverage results +*.coverage +*.coveragexml + +# NCrunch +_NCrunch_* +.*crunch*.local.xml +nCrunchTemp_* + +# MightyMoose +*.mm.* +AutoTest.Net/ + +# Web workbench (sass) +.sass-cache/ + +# Installshield output folder +[Ee]xpress/ + +# DocProject is a documentation generator add-in +DocProject/buildhelp/ +DocProject/Help/*.HxT +DocProject/Help/*.HxC +DocProject/Help/*.hhc +DocProject/Help/*.hhk +DocProject/Help/*.hhp +DocProject/Help/Html2 +DocProject/Help/html + +# Click-Once directory +publish/ + +# Publish Web Output +*.[Pp]ublish.xml +*.azurePubxml +# Note: Comment the next line if you want to checkin your web deploy settings, +# but database connection strings (with potential passwords) will be unencrypted +*.pubxml +*.publishproj + +# Microsoft Azure Web App publish settings. Comment the next line if you want to +# checkin your Azure Web App publish settings, but sensitive information contained +# in these scripts will be unencrypted +PublishScripts/ + +# NuGet Packages +*.nupkg +# NuGet Symbol Packages +*.snupkg +# The packages folder can be ignored because of Package Restore +**/[Pp]ackages/* +# except build/, which is used as an MSBuild target. +!**/[Pp]ackages/build/ +# Uncomment if necessary however generally it will be regenerated when needed +#!**/[Pp]ackages/repositories.config +# NuGet v3's project.json files produces more ignorable files +*.nuget.props +*.nuget.targets + +# Microsoft Azure Build Output +csx/ +*.build.csdef + +# Microsoft Azure Emulator +ecf/ +rcf/ + +# Windows Store app package directories and files +AppPackages/ +BundleArtifacts/ +Package.StoreAssociation.xml +_pkginfo.txt +*.appx +*.appxbundle +*.appxupload + +# Visual Studio cache files +# files ending in .cache can be ignored +*.[Cc]ache +# but keep track of directories ending in .cache +!?*.[Cc]ache/ + +# Others +ClientBin/ +~$* +*~ +*.dbmdl +*.dbproj.schemaview +*.jfm +*.pfx +*.publishsettings +orleans.codegen.cs + +# Including strong name files can present a security risk +# (https://github.com/github/gitignore/pull/2483#issue-259490424) +#*.snk + +# Since there are multiple workflows, uncomment next line to ignore bower_components +# (https://github.com/github/gitignore/pull/1529#issuecomment-104372622) +#bower_components/ + +# RIA/Silverlight projects +Generated_Code/ + +# Backup & report files from converting an old project file +# to a newer Visual Studio version. Backup files are not needed, +# because we have git ;-) +_UpgradeReport_Files/ +Backup*/ +UpgradeLog*.XML +UpgradeLog*.htm +ServiceFabricBackup/ +*.rptproj.bak + +# SQL Server files +*.mdf +*.ldf +*.ndf + +# Business Intelligence projects +*.rdl.data +*.bim.layout +*.bim_*.settings +*.rptproj.rsuser +*- [Bb]ackup.rdl +*- [Bb]ackup ([0-9]).rdl +*- [Bb]ackup ([0-9][0-9]).rdl + +# Microsoft Fakes +FakesAssemblies/ + +# GhostDoc plugin setting file +*.GhostDoc.xml + +# Node.js Tools for Visual Studio +.ntvs_analysis.dat +node_modules/ + +# Visual Studio 6 build log +*.plg + +# Visual Studio 6 workspace options file +*.opt + +# Visual Studio 6 auto-generated workspace file (contains which files were open etc.) +*.vbw + +# Visual Studio LightSwitch build output +**/*.HTMLClient/GeneratedArtifacts +**/*.DesktopClient/GeneratedArtifacts +**/*.DesktopClient/ModelManifest.xml +**/*.Server/GeneratedArtifacts +**/*.Server/ModelManifest.xml +_Pvt_Extensions + +# Paket dependency manager +.paket/paket.exe +paket-files/ + +# FAKE - F# Make +.fake/ + +# CodeRush personal settings +.cr/personal + +# Python Tools for Visual Studio (PTVS) +__pycache__/ +*.pyc + +# Cake - Uncomment if you are using it +# tools/** +# !tools/packages.config + +# Tabs Studio +*.tss + +# Telerik's JustMock configuration file +*.jmconfig + +# BizTalk build output +*.btp.cs +*.btm.cs +*.odx.cs +*.xsd.cs + +# OpenCover UI analysis results +OpenCover/ + +# Azure Stream Analytics local run output +ASALocalRun/ + +# MSBuild Binary and Structured Log +*.binlog + +# NVidia Nsight GPU debugger configuration file +*.nvuser + +# MFractors (Xamarin productivity tool) working folder +.mfractor/ + +# Local History for Visual Studio +.localhistory/ + +# BeatPulse healthcheck temp database +healthchecksdb + +# Backup folder for Package Reference Convert tool in Visual Studio 2017 +MigrationBackup/ + +# Ionide (cross platform F# VS Code tools) working folder +.ionide/ + +# Fody - auto-generated XML schema +FodyWeavers.xsd + +### Windows template +# Windows thumbnail cache files +Thumbs.db +Thumbs.db:encryptable +ehthumbs.db +ehthumbs_vista.db + +# Dump file +*.stackdump + +# Folder config file +[Dd]esktop.ini + +# Recycle Bin used on file shares +$RECYCLE.BIN/ + +# Windows Installer files +*.cab +*.msi +*.msix +*.msm +*.msp + +# Windows shortcuts +*.lnk + diff --git a/message.go b/message.go index 1933681..c1316eb 100644 --- a/message.go +++ b/message.go @@ -1 +1,17 @@ package go_cbox_scraper + +import ( + "fmt" + "time" +) + +type CBoxMessage struct { + MessageID int + DateTime time.Time + Username string + Message string +} + +func (m *CBoxMessage) String() string { + return fmt.Sprintf("%d - %s - %s - %s", m.MessageID, m.DateTime.Format(time.Stamp), m.Username, m.Message) +} diff --git a/page.go b/page.go index 1933681..1c49714 100644 --- a/page.go +++ b/page.go @@ -1 +1,251 @@ package go_cbox_scraper + +import ( + "errors" + "fmt" + "log" + "net/http" + "net/url" + "regexp" + "strconv" + "time" + + "github.com/PuerkitoBio/goquery" +) + +var ( + digitRegex *regexp.Regexp +) + +func init() { + digitRegex = regexp.MustCompile(`\d+`) +} + +type CBoxPage struct { + Messages map[int]*CBoxMessage + CanPaginate bool + + index int + previousIndex int + + canPaginatePrevious bool + canPaginateNext bool + previousString string + nextString string + + previousPage int + nextPage int + + smallestID int + largestID int + + section string + + CBoxServerInfo +} + +func NewCBoxPage(cbxInfo CBoxServerInfo) *CBoxPage { + return &CBoxPage{ + Messages: make(map[int]*CBoxMessage), + CanPaginate: false, + index: -1, + previousIndex: -1, + canPaginatePrevious: false, + canPaginateNext: false, + previousString: "", + nextString: "", + previousPage: -1, + nextPage: -1, + smallestID: -1, + largestID: -1, + section: "", + CBoxServerInfo: cbxInfo, + } +} + +func (p *CBoxPage) SmallestID() int { + return p.smallestID +} + +func (p *CBoxPage) LargestID() int { + return p.largestID +} + +func (p *CBoxPage) FetchMain() error { + cboxURL := p.buildCboxURL("main") + + request, err := newHTTPRequest(cboxURL) + if err != nil { + return err + } + + document, err := p.request(request) + if err != nil { + return err + } + + p.parsePage(document) + + return nil +} + +func (p *CBoxPage) FetchPrevious() error { + if !p.canPaginatePrevious { + return errors.New("can not paginate previous") + } + + cboxURL := p.buildCboxURL("archive") + cboxURL.RawQuery = p.previousString + + request, err := newHTTPRequest(cboxURL) + if err != nil { + return nil + } + + document, err := p.request(request) + if err != nil { + return err + } + + p.parsePage(document) + + return nil +} + +func (p *CBoxPage) FetchNext() error { + if !p.canPaginateNext { + return errors.New("can not paginate next") + } + + cboxURL := p.buildCboxURL("archive") + cboxURL.RawQuery = p.nextString + + request, err := newHTTPRequest(cboxURL) + if err != nil { + return err + } + + document, err := p.request(request) + if err != nil { + return err + } + + p.parsePage(document) + + return nil +} + +func (p *CBoxPage) request(request *http.Request) (*goquery.Document, error) { + if p.Debug { + log.Println(request.URL.String()) + } + + httpClient := &http.Client{ + Timeout: 30 * time.Second, + } + + // Make request + response, err := httpClient.Do(request) + if err != nil { + return nil, err + } + defer response.Body.Close() + + // Create a goquery document from the HTTP response + document, err := goquery.NewDocumentFromReader(response.Body) + if err != nil { + return nil, err + } + + return document, nil +} + +func (p *CBoxPage) buildCboxURL(section string) *url.URL { + // format := "https://www%d.cbox.ws/box/?boxid=%d&boxtag=%s&sec=%s" + u := &url.URL{ + Host: fmt.Sprintf("www%d.cbox.ws", p.WebHostID), + Scheme: "https", + Path: "/box/", + } + q := u.Query() + q.Set("boxid", strconv.Itoa(p.BoxID)) + q.Set("boxtag", p.BoxTag) + q.Set("sec", section) + u.RawQuery = q.Encode() + return u +} + +func (p *CBoxPage) parsePage(document *goquery.Document) { + document.Find(".msg").Each(func(index int, element *goquery.Selection) { + messageIDString, exists := element.Attr("id") + if exists { + messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString)) + if p.smallestID == -1 || messageIDInt < p.smallestID { + p.smallestID = messageIDInt + } + if p.largestID == -1 || messageIDInt > p.largestID { + p.largestID = messageIDInt + } + p.Messages[messageIDInt] = p.parseMessage(messageIDInt, element) + } + }) + + // can paginate backwards on main page + _, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href") + if p.canPaginatePrevious { + cboxURL := p.buildCboxURL("archive") + q := cboxURL.Query() + q.Set("i", strconv.Itoa(p.smallestID)) + p.previousString = q.Encode() + } else { + // in an archive page + previousString, canPrevious := document.Find("td[align='left'] a").Attr("href") + if canPrevious { + p.canPaginatePrevious = true + p.previousString = previousString[3:] + } + + nextString, canNext := document.Find("td[align='right'] a").Attr("href") + if canNext { + p.canPaginateNext = true + p.nextString = nextString[3:] + } + } +} + +func (p *CBoxPage) parseMessage(messageID int, element *goquery.Selection) *CBoxMessage { + message := CBoxMessage{ + MessageID: messageID, + } + + datetimeElement := element.Find("div").Text() + if datetimeElement != "" { + message.DateTime, _ = time.Parse(CboxDatetimeFormat, datetimeElement) + } + element.Find("div").Remove() + + name := element.Find("b.nme").Text() + if name != "" { + message.Username = name + } + element.Find("b.nme").Remove() + + // trim the first 2 characters, colon and space + message.Message = element.Text()[2:] + + return &message +} + +func newHTTPRequest(url *url.URL) (*http.Request, error) { + // Create and modify HTTP request before sending + request, err := http.NewRequest("GET", url.String(), nil) + if err != nil { + return nil, err + } + request.Header.Set("Accept", "image/gif, image/x-xbitmap, image/jpeg, image/pjpeg, image/xbm, */* ") + request.Header.Set("Accept-Language", "en") + request.Header.Set("Connection", "Keep-Alive") + request.Header.Set("User-Agent", "Mozilla/4.0 (compatible; MSIE 4.01; AOL 4.0; Windows 98)") + + return request, nil +} diff --git a/scraper.go b/scraper.go index 1933681..be6035c 100644 --- a/scraper.go +++ b/scraper.go @@ -1 +1,160 @@ package go_cbox_scraper + +import ( + "encoding/gob" + "log" + "os" + "time" +) + +const CboxDatetimeFormat = "_2 Jan 06, 03:04 PM" + +type CBoxScraper struct { + SmallestMessageID int + LargestMessageID int + Messages map[int]*CBoxMessage + CBoxServerInfo +} + +func NewScraper(cboxServerInfo CBoxServerInfo, smallestID int, largestID int) *CBoxScraper { + return &CBoxScraper{ + SmallestMessageID: smallestID, + LargestMessageID: largestID, + CBoxServerInfo: cboxServerInfo, + Messages: make(map[int]*CBoxMessage), + } +} + +func (s *CBoxScraper) sleep() { + if s.Debug { + log.Println("Sleeping 10 seconds...") + } + time.Sleep(10 * time.Second) +} + +func (s *CBoxScraper) Scrape(updatesOnly bool) error { + if s.Debug { + log.Println("Scraper Started...") + } + + page := NewCBoxPage(s.CBoxServerInfo) + + err := page.FetchMain() + if err != nil { + return err + } + + if s.Debug { + log.Printf("Main fetched, scraped %d messages\n", len(page.Messages)) + log.Printf("\tpage smallestID: %d - scraper smallestID: %d\n", page.SmallestID(), s.SmallestMessageID) + log.Printf("\tpage largestID: %d - scraper LargestID: %d\n", page.LargestID(), s.LargestMessageID) + } + + if updatesOnly && s.LargestMessageID < page.SmallestID() { + if s.Debug { + log.Println("Scraper: Case 3 - lots of new messages") + } + s.sleep() + for s.LargestMessageID < page.smallestID { + err := page.FetchPrevious() + if err != nil { + break + } + s.sleep() + } + s.merge(page.Messages) + } else if updatesOnly && s.LargestMessageID < page.LargestID() { + log.Println("Scraper: Case 2 - Some new Messages") + // merge what we retrieved + s.merge(page.Messages) + } else if !updatesOnly { + log.Println("Scraper: Case 4 - fetch all") + s.sleep() + for { + err := page.FetchPrevious() + if err != nil { + break + } + s.sleep() + } + s.merge(page.Messages) + } else { + log.Println("Scraper: Case 0 - no update") + } + + if s.Debug { + log.Printf("Archives fetched, scraped %d messages - page smallestID: %d\n", len(page.Messages), page.SmallestID()) + } + + return nil +} + +func (s *CBoxScraper) merge(input map[int]*CBoxMessage) { + for k,v := range input { + s.Messages[k] = v + } + s.updateIndices() +} + +func (s *CBoxScraper) updateIndices() { + s.SmallestMessageID = -1 + s.LargestMessageID = -1 + for k, _ := range s.Messages { + if s.SmallestMessageID == -1 || s.SmallestMessageID > k { + s.SmallestMessageID = k + } + if s.LargestMessageID == -1 || s.LargestMessageID < k { + s.LargestMessageID = k + } + } +} + +func (s *CBoxScraper) Save(path string) error { + flags := os.O_TRUNC | os.O_RDWR | os.O_EXCL + file, err := os.Stat(path) + if file == nil { + flags |= os.O_CREATE + } + + dataFile, err := os.OpenFile(path, flags, 0644) + if err != nil { + return err + } + + defer dataFile.Close() + + dataEncoder := gob.NewEncoder(dataFile) + err = dataEncoder.Encode(s) + + if err != nil { + return err + } + + return nil +} + +func (s *CBoxScraper) Load(path string) error { + file, err := os.Stat(path) + if file == nil { + err = s.Save(path) + } + if err != nil { + return err + } + + dataFile, err := os.OpenFile(path, os.O_RDWR|os.O_EXCL, 0644) + if err != nil { + return err + } + + defer dataFile.Close() + + dataDecoder := gob.NewDecoder(dataFile) + err = dataDecoder.Decode(s) + + if err != nil { + return err + } + + return nil +} diff --git a/serverInfo.go b/serverInfo.go index 1933681..9439e1b 100644 --- a/serverInfo.go +++ b/serverInfo.go @@ -1 +1,8 @@ package go_cbox_scraper + +type CBoxServerInfo struct { + WebHostID int + BoxID int + BoxTag string + Debug bool +}