add functionality

This commit is contained in:
Amrit Panesar committed 2020-10-29 15:17:33 -07:00
1 parent 11cee3336d
commit b87ac1bb0e
5 files changed
+992

No files matched your search

+560
View File
@@ -1 +1,561 @@
# Created by .ignore support plugin (hsz.mobi)
### Go template
# Binaries for programs and plugins
*.exe
*.exe~
*.dll
*.so
*.dylib
# Test binary, built with `go test -c`
*.test
# Output of the go coverage tool, specifically when used with LiteIDE
*.out
# Dependency directories (remove the comment below to include it)
# vendor/
### Linux template
*~
# temporary files which can be created if a process still has a handle open of a deleted file
.fuse_hidden*
# KDE directory preferences
.directory
# Linux trash folder which might appear on any partition or disk
.Trash-*
# .nfs files are created when an open file is removed but is still being accessed
.nfs*
### macOS template
# General
.DS_Store
.AppleDouble
.LSOverride
# Icon must end with two \r
Icon
# Thumbnails
._*
# Files that might appear in the root of a volume
.DocumentRevisions-V100
.fseventsd
.Spotlight-V100
.TemporaryItems
.Trashes
.VolumeIcon.icns
.com.apple.timemachine.donotpresent
# Directories potentially created on remote AFP share
.AppleDB
.AppleDesktop
Network Trash Folder
Temporary Items
.apdisk
### SublimeText template
# Cache files for Sublime Text
*.tmlanguage.cache
*.tmPreferences.cache
*.stTheme.cache
# Workspace files are user-specific
*.sublime-workspace
# Project files should be checked into the repository, unless a significant
# proportion of contributors will probably not be using Sublime Text
# *.sublime-project
# SFTP configuration file
sftp-config.json
sftp-config-alt*.json
# Package control specific files
Package Control.last-run
Package Control.ca-list
Package Control.ca-bundle
Package Control.system-ca-bundle
Package Control.cache/
Package Control.ca-certs/
Package Control.merged-ca-bundle
Package Control.user-ca-bundle
oscrypto-ca-bundle.crt
bh_unicode_properties.cache
# Sublime-github package stores a github token in this file
# https://packagecontrol.io/packages/sublime-github
GitHub.sublime-settings
### JetBrains template
# Covers JetBrains IDEs: IntelliJ, RubyMine, PhpStorm, AppCode, PyCharm, CLion, Android Studio, WebStorm and Rider
# Reference: https://intellij-support.jetbrains.com/hc/en-us/articles/206544839
# User-specific stuff
.idea/**/workspace.xml
.idea/**/tasks.xml
.idea/**/usage.statistics.xml
.idea/**/dictionaries
.idea/**/shelf
# Generated files
.idea/**/contentModel.xml
# Sensitive or high-churn files
.idea/**/dataSources/
.idea/**/dataSources.ids
.idea/**/dataSources.local.xml
.idea/**/sqlDataSources.xml
.idea/**/dynamic.xml
.idea/**/uiDesigner.xml
.idea/**/dbnavigator.xml
# Gradle
.idea/**/gradle.xml
.idea/**/libraries
# Gradle and Maven with auto-import
# When using Gradle or Maven with auto-import, you should exclude module files,
# since they will be recreated, and may cause churn. Uncomment if using
# auto-import.
# .idea/artifacts
# .idea/compiler.xml
# .idea/jarRepositories.xml
# .idea/modules.xml
# .idea/*.iml
# .idea/modules
# *.iml
# *.ipr
# CMake
cmake-build-*/
# Mongo Explorer plugin
.idea/**/mongoSettings.xml
# File-based project format
*.iws
# IntelliJ
out/
# mpeltonen/sbt-idea plugin
.idea_modules/
# JIRA plugin
atlassian-ide-plugin.xml
# Cursive Clojure plugin
.idea/replstate.xml
# Crashlytics plugin (for Android Studio and IntelliJ)
com_crashlytics_export_strings.xml
crashlytics.properties
crashlytics-build.properties
fabric.properties
# Editor-based Rest Client
.idea/httpRequests
# Android studio 3.1+ serialized cache file
.idea/caches/build_file_checksums.ser
### NotepadPP template
# Notepad++ backups #
*.bak
### VisualStudio template
## Ignore Visual Studio temporary files, build results, and
## files generated by popular Visual Studio add-ons.
##
## Get latest from https://github.com/github/gitignore/blob/master/VisualStudio.gitignore
# User-specific files
*.rsuser
*.suo
*.user
*.userosscache
*.sln.docstates
# User-specific files (MonoDevelop/Xamarin Studio)
*.userprefs
# Mono auto generated files
mono_crash.*
# Build results
[Dd]ebug/
[Dd]ebugPublic/
[Rr]elease/
[Rr]eleases/
x64/
x86/
[Ww][Ii][Nn]32/
[Aa][Rr][Mm]/
[Aa][Rr][Mm]64/
bld/
[Bb]in/
[Oo]bj/
[Ll]og/
[Ll]ogs/
# Visual Studio 2015/2017 cache/options directory
.vs/
# Uncomment if you have tasks that create the project's static files in wwwroot
#wwwroot/
# Visual Studio 2017 auto generated files
Generated\ Files/
# MSTest test Results
[Tt]est[Rr]esult*/
[Bb]uild[Ll]og.*
# NUnit
*.VisualState.xml
TestResult.xml
nunit-*.xml
# Build Results of an ATL Project
[Dd]ebugPS/
[Rr]eleasePS/
dlldata.c
# Benchmark Results
BenchmarkDotNet.Artifacts/
# .NET Core
project.lock.json
project.fragment.lock.json
artifacts/
# ASP.NET Scaffolding
ScaffoldingReadMe.txt
# StyleCop
StyleCopReport.xml
# Files built by Visual Studio
*_i.c
*_p.c
*_h.h
*.ilk
*.meta
*.obj
*.iobj
*.pch
*.pdb
*.ipdb
*.pgc
*.pgd
*.rsp
*.sbr
*.tlb
*.tli
*.tlh
*.tmp
*.tmp_proj
*_wpftmp.csproj
*.log
*.vspscc
*.vssscc
.builds
*.pidb
*.svclog
*.scc
# Chutzpah Test files
_Chutzpah*
# Visual C++ cache files
ipch/
*.aps
*.ncb
*.opendb
*.opensdf
*.sdf
*.cachefile
*.VC.db
*.VC.VC.opendb
# Visual Studio profiler
*.psess
*.vsp
*.vspx
*.sap
# Visual Studio Trace Files
*.e2e
# TFS 2012 Local Workspace
$tf/
# Guidance Automation Toolkit
*.gpState
# ReSharper is a .NET coding add-in
_ReSharper*/
*.[Rr]e[Ss]harper
*.DotSettings.user
# TeamCity is a build add-in
_TeamCity*
# DotCover is a Code Coverage Tool
*.dotCover
# AxoCover is a Code Coverage Tool
.axoCover/*
!.axoCover/settings.json
# Coverlet is a free, cross platform Code Coverage Tool
coverage*.json
coverage*.xml
coverage*.info
# Visual Studio code coverage results
*.coverage
*.coveragexml
# NCrunch
_NCrunch_*
.*crunch*.local.xml
nCrunchTemp_*
# MightyMoose
*.mm.*
AutoTest.Net/
# Web workbench (sass)
.sass-cache/
# Installshield output folder
[Ee]xpress/
# DocProject is a documentation generator add-in
DocProject/buildhelp/
DocProject/Help/*.HxT
DocProject/Help/*.HxC
DocProject/Help/*.hhc
DocProject/Help/*.hhk
DocProject/Help/*.hhp
DocProject/Help/Html2
DocProject/Help/html
# Click-Once directory
publish/
# Publish Web Output
*.[Pp]ublish.xml
*.azurePubxml
# Note: Comment the next line if you want to checkin your web deploy settings,
# but database connection strings (with potential passwords) will be unencrypted
*.pubxml
*.publishproj
# Microsoft Azure Web App publish settings. Comment the next line if you want to
# checkin your Azure Web App publish settings, but sensitive information contained
# in these scripts will be unencrypted
PublishScripts/
# NuGet Packages
*.nupkg
# NuGet Symbol Packages
*.snupkg
# The packages folder can be ignored because of Package Restore
**/[Pp]ackages/*
# except build/, which is used as an MSBuild target.
!**/[Pp]ackages/build/
# Uncomment if necessary however generally it will be regenerated when needed
#!**/[Pp]ackages/repositories.config
# NuGet v3's project.json files produces more ignorable files
*.nuget.props
*.nuget.targets
# Microsoft Azure Build Output
csx/
*.build.csdef
# Microsoft Azure Emulator
ecf/
rcf/
# Windows Store app package directories and files
AppPackages/
BundleArtifacts/
Package.StoreAssociation.xml
_pkginfo.txt
*.appx
*.appxbundle
*.appxupload
# Visual Studio cache files
# files ending in .cache can be ignored
*.[Cc]ache
# but keep track of directories ending in .cache
!?*.[Cc]ache/
# Others
ClientBin/
~$*
*~
*.dbmdl
*.dbproj.schemaview
*.jfm
*.pfx
*.publishsettings
orleans.codegen.cs
# Including strong name files can present a security risk
# (https://github.com/github/gitignore/pull/2483#issue-259490424)
#*.snk
# Since there are multiple workflows, uncomment next line to ignore bower_components
# (https://github.com/github/gitignore/pull/1529#issuecomment-104372622)
#bower_components/
# RIA/Silverlight projects
Generated_Code/
# Backup & report files from converting an old project file
# to a newer Visual Studio version. Backup files are not needed,
# because we have git ;-)
_UpgradeReport_Files/
Backup*/
UpgradeLog*.XML
UpgradeLog*.htm
ServiceFabricBackup/
*.rptproj.bak
# SQL Server files
*.mdf
*.ldf
*.ndf
# Business Intelligence projects
*.rdl.data
*.bim.layout
*.bim_*.settings
*.rptproj.rsuser
*- [Bb]ackup.rdl
*- [Bb]ackup ([0-9]).rdl
*- [Bb]ackup ([0-9][0-9]).rdl
# Microsoft Fakes
FakesAssemblies/
# GhostDoc plugin setting file
*.GhostDoc.xml
# Node.js Tools for Visual Studio
.ntvs_analysis.dat
node_modules/
# Visual Studio 6 build log
*.plg
# Visual Studio 6 workspace options file
*.opt
# Visual Studio 6 auto-generated workspace file (contains which files were open etc.)
*.vbw
# Visual Studio LightSwitch build output
**/*.HTMLClient/GeneratedArtifacts
**/*.DesktopClient/GeneratedArtifacts
**/*.DesktopClient/ModelManifest.xml
**/*.Server/GeneratedArtifacts
**/*.Server/ModelManifest.xml
_Pvt_Extensions
# Paket dependency manager
.paket/paket.exe
paket-files/
# FAKE - F# Make
.fake/
# CodeRush personal settings
.cr/personal
# Python Tools for Visual Studio (PTVS)
__pycache__/
*.pyc
# Cake - Uncomment if you are using it
# tools/**
# !tools/packages.config
# Tabs Studio
*.tss
# Telerik's JustMock configuration file
*.jmconfig
# BizTalk build output
*.btp.cs
*.btm.cs
*.odx.cs
*.xsd.cs
# OpenCover UI analysis results
OpenCover/
# Azure Stream Analytics local run output
ASALocalRun/
# MSBuild Binary and Structured Log
*.binlog
# NVidia Nsight GPU debugger configuration file
*.nvuser
# MFractors (Xamarin productivity tool) working folder
.mfractor/
# Local History for Visual Studio
.localhistory/
# BeatPulse healthcheck temp database
healthchecksdb
# Backup folder for Package Reference Convert tool in Visual Studio 2017
MigrationBackup/
# Ionide (cross platform F# VS Code tools) working folder
.ionide/
# Fody - auto-generated XML schema
FodyWeavers.xsd
### Windows template
# Windows thumbnail cache files
Thumbs.db
Thumbs.db:encryptable
ehthumbs.db
ehthumbs_vista.db
# Dump file
*.stackdump
# Folder config file
[Dd]esktop.ini
# Recycle Bin used on file shares
$RECYCLE.BIN/
# Windows Installer files
*.cab
*.msi
*.msix
*.msm
*.msp
# Windows shortcuts
*.lnk
+16
View File
@@ -1 +1,17 @@
package go_cbox_scraper
import (
"fmt"
"time"
)
type CBoxMessage struct {
MessageID int
DateTime time.Time
Username string
Message string
}
func (m *CBoxMessage) String() string {
return fmt.Sprintf("%d - %s - %s - %s", m.MessageID, m.DateTime.Format(time.Stamp), m.Username, m.Message)
}
+250
View File
@@ -1 +1,251 @@
package go_cbox_scraper
import (
"errors"
"fmt"
"log"
"net/http"
"net/url"
"regexp"
"strconv"
"time"
"github.com/PuerkitoBio/goquery"
)
var (
digitRegex *regexp.Regexp
)
func init() {
digitRegex = regexp.MustCompile(`\d+`)
}
type CBoxPage struct {
Messages map[int]*CBoxMessage
CanPaginate bool
index int
previousIndex int
canPaginatePrevious bool
canPaginateNext bool
previousString string
nextString string
previousPage int
nextPage int
smallestID int
largestID int
section string
CBoxServerInfo
}
func NewCBoxPage(cbxInfo CBoxServerInfo) *CBoxPage {
return &CBoxPage{
Messages: make(map[int]*CBoxMessage),
CanPaginate: false,
index: -1,
previousIndex: -1,
canPaginatePrevious: false,
canPaginateNext: false,
previousString: "",
nextString: "",
previousPage: -1,
nextPage: -1,
smallestID: -1,
largestID: -1,
section: "",
CBoxServerInfo: cbxInfo,
}
}
func (p *CBoxPage) SmallestID() int {
return p.smallestID
}
func (p *CBoxPage) LargestID() int {
return p.largestID
}
func (p *CBoxPage) FetchMain() error {
cboxURL := p.buildCboxURL("main")
request, err := newHTTPRequest(cboxURL)
if err != nil {
return err
}
document, err := p.request(request)
if err != nil {
return err
}
p.parsePage(document)
return nil
}
func (p *CBoxPage) FetchPrevious() error {
if !p.canPaginatePrevious {
return errors.New("can not paginate previous")
}
cboxURL := p.buildCboxURL("archive")
cboxURL.RawQuery = p.previousString
request, err := newHTTPRequest(cboxURL)
if err != nil {
return nil
}
document, err := p.request(request)
if err != nil {
return err
}
p.parsePage(document)
return nil
}
func (p *CBoxPage) FetchNext() error {
if !p.canPaginateNext {
return errors.New("can not paginate next")
}
cboxURL := p.buildCboxURL("archive")
cboxURL.RawQuery = p.nextString
request, err := newHTTPRequest(cboxURL)
if err != nil {
return err
}
document, err := p.request(request)
if err != nil {
return err
}
p.parsePage(document)
return nil
}
func (p *CBoxPage) request(request *http.Request) (*goquery.Document, error) {
if p.Debug {
log.Println(request.URL.String())
}
httpClient := &http.Client{
Timeout: 30 * time.Second,
}
// Make request
response, err := httpClient.Do(request)
if err != nil {
return nil, err
}
defer response.Body.Close()
// Create a goquery document from the HTTP response
document, err := goquery.NewDocumentFromReader(response.Body)
if err != nil {
return nil, err
}
return document, nil
}
func (p *CBoxPage) buildCboxURL(section string) *url.URL {
// format := "https://www%d.cbox.ws/box/?boxid=%d&boxtag=%s&sec=%s"
u := &url.URL{
Host: fmt.Sprintf("www%d.cbox.ws", p.WebHostID),
Scheme: "https",
Path: "/box/",
}
q := u.Query()
q.Set("boxid", strconv.Itoa(p.BoxID))
q.Set("boxtag", p.BoxTag)
q.Set("sec", section)
u.RawQuery = q.Encode()
return u
}
func (p *CBoxPage) parsePage(document *goquery.Document) {
document.Find(".msg").Each(func(index int, element *goquery.Selection) {
messageIDString, exists := element.Attr("id")
if exists {
messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString))
if p.smallestID == -1 || messageIDInt < p.smallestID {
p.smallestID = messageIDInt
}
if p.largestID == -1 || messageIDInt > p.largestID {
p.largestID = messageIDInt
}
p.Messages[messageIDInt] = p.parseMessage(messageIDInt, element)
}
})
// can paginate backwards on main page
_, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href")
if p.canPaginatePrevious {
cboxURL := p.buildCboxURL("archive")
q := cboxURL.Query()
q.Set("i", strconv.Itoa(p.smallestID))
p.previousString = q.Encode()
} else {
// in an archive page
previousString, canPrevious := document.Find("td[align='left'] a").Attr("href")
if canPrevious {
p.canPaginatePrevious = true
p.previousString = previousString[3:]
}
nextString, canNext := document.Find("td[align='right'] a").Attr("href")
if canNext {
p.canPaginateNext = true
p.nextString = nextString[3:]
}
}
}
func (p *CBoxPage) parseMessage(messageID int, element *goquery.Selection) *CBoxMessage {
message := CBoxMessage{
MessageID: messageID,
}
datetimeElement := element.Find("div").Text()
if datetimeElement != "" {
message.DateTime, _ = time.Parse(CboxDatetimeFormat, datetimeElement)
}
element.Find("div").Remove()
name := element.Find("b.nme").Text()
if name != "" {
message.Username = name
}
element.Find("b.nme").Remove()
// trim the first 2 characters, colon and space
message.Message = element.Text()[2:]
return &message
}
func newHTTPRequest(url *url.URL) (*http.Request, error) {
// Create and modify HTTP request before sending
request, err := http.NewRequest("GET", url.String(), nil)
if err != nil {
return nil, err
}
request.Header.Set("Accept", "image/gif, image/x-xbitmap, image/jpeg, image/pjpeg, image/xbm, */* ")
request.Header.Set("Accept-Language", "en")
request.Header.Set("Connection", "Keep-Alive")
request.Header.Set("User-Agent", "Mozilla/4.0 (compatible; MSIE 4.01; AOL 4.0; Windows 98)")
return request, nil
}
+159
View File
@@ -1 +1,160 @@
package go_cbox_scraper
import (
"encoding/gob"
"log"
"os"
"time"
)
const CboxDatetimeFormat = "_2 Jan 06, 03:04 PM"
type CBoxScraper struct {
SmallestMessageID int
LargestMessageID int
Messages map[int]*CBoxMessage
CBoxServerInfo
}
func NewScraper(cboxServerInfo CBoxServerInfo, smallestID int, largestID int) *CBoxScraper {
return &CBoxScraper{
SmallestMessageID: smallestID,
LargestMessageID: largestID,
CBoxServerInfo: cboxServerInfo,
Messages: make(map[int]*CBoxMessage),
}
}
func (s *CBoxScraper) sleep() {
if s.Debug {
log.Println("Sleeping 10 seconds...")
}
time.Sleep(10 * time.Second)
}
func (s *CBoxScraper) Scrape(updatesOnly bool) error {
if s.Debug {
log.Println("Scraper Started...")
}
page := NewCBoxPage(s.CBoxServerInfo)
err := page.FetchMain()
if err != nil {
return err
}
if s.Debug {
log.Printf("Main fetched, scraped %d messages\n", len(page.Messages))
log.Printf("\tpage smallestID: %d - scraper smallestID: %d\n", page.SmallestID(), s.SmallestMessageID)
log.Printf("\tpage largestID: %d - scraper LargestID: %d\n", page.LargestID(), s.LargestMessageID)
}
if updatesOnly && s.LargestMessageID < page.SmallestID() {
if s.Debug {
log.Println("Scraper: Case 3 - lots of new messages")
}
s.sleep()
for s.LargestMessageID < page.smallestID {
err := page.FetchPrevious()
if err != nil {
break
}
s.sleep()
}
s.merge(page.Messages)
} else if updatesOnly && s.LargestMessageID < page.LargestID() {
log.Println("Scraper: Case 2 - Some new Messages")
// merge what we retrieved
s.merge(page.Messages)
} else if !updatesOnly {
log.Println("Scraper: Case 4 - fetch all")
s.sleep()
for {
err := page.FetchPrevious()
if err != nil {
break
}
s.sleep()
}
s.merge(page.Messages)
} else {
log.Println("Scraper: Case 0 - no update")
}
if s.Debug {
log.Printf("Archives fetched, scraped %d messages - page smallestID: %d\n", len(page.Messages), page.SmallestID())
}
return nil
}
func (s *CBoxScraper) merge(input map[int]*CBoxMessage) {
for k,v := range input {
s.Messages[k] = v
}
s.updateIndices()
}
func (s *CBoxScraper) updateIndices() {
s.SmallestMessageID = -1
s.LargestMessageID = -1
for k, _ := range s.Messages {
if s.SmallestMessageID == -1 || s.SmallestMessageID > k {
s.SmallestMessageID = k
}
if s.LargestMessageID == -1 || s.LargestMessageID < k {
s.LargestMessageID = k
}
}
}
func (s *CBoxScraper) Save(path string) error {
flags := os.O_TRUNC | os.O_RDWR | os.O_EXCL
file, err := os.Stat(path)
if file == nil {
flags |= os.O_CREATE
}
dataFile, err := os.OpenFile(path, flags, 0644)
if err != nil {
return err
}
defer dataFile.Close()
dataEncoder := gob.NewEncoder(dataFile)
err = dataEncoder.Encode(s)
if err != nil {
return err
}
return nil
}
func (s *CBoxScraper) Load(path string) error {
file, err := os.Stat(path)
if file == nil {
err = s.Save(path)
}
if err != nil {
return err
}
dataFile, err := os.OpenFile(path, os.O_RDWR|os.O_EXCL, 0644)
if err != nil {
return err
}
defer dataFile.Close()
dataDecoder := gob.NewDecoder(dataFile)
err = dataDecoder.Decode(s)
if err != nil {
return err
}
return nil
}
+7
View File
@@ -1 +1,8 @@
package go_cbox_scraper
type CBoxServerInfo struct {
WebHostID int
BoxID int
BoxTag string
Debug bool
}