update scraping logic to handle new style cbox pages and new unified dateformat

This commit is contained in:
Neo-Desktop committed 2021-10-04 09:00:49 -07:00
1 parent 7ebc9d5acc
commit a7d0094a08
5 files changed
+183 -57

No files matched your search

+3 -1
View File
@@ -12,6 +12,8 @@ type Message struct {
Message string
}
const DisplayDatetimeFormat = "2006-01-02 03:04PM"
func (m *Message) String() string {
return fmt.Sprintf("#%d [%s] <%s> %s", m.MessageID, m.DateTime.Format(DatetimeFormat), m.Username, m.Message)
return fmt.Sprintf("#%d [%s] <%s> %s", m.MessageID, m.DateTime.Format(DisplayDatetimeFormat), m.Username, m.Message)
}
+20 -55
View File
@@ -3,6 +3,7 @@ package go_cbox_scraper
import (
"errors"
"fmt"
"log"
"net/http"
"net/url"
"regexp"
@@ -14,10 +15,17 @@ import (
var (
digitRegex *regexp.Regexp
chicago *time.Location
)
func init() {
var err error
digitRegex = regexp.MustCompile(`\d+`)
chicago, err = time.LoadLocation("America/Chicago")
if err != nil {
panic("Unable to load timezone offset for America/Chicago from local timezone database")
}
}
type CBoxPage struct {
@@ -40,6 +48,8 @@ type CBoxPage struct {
section string
isOld bool
ServerInfo
}
@@ -58,6 +68,7 @@ func NewCBoxPage(cbxInfo ServerInfo) *CBoxPage {
smallestID: -1,
largestID: -1,
section: "",
isOld: true,
ServerInfo: cbxInfo,
}
}
@@ -173,64 +184,18 @@ func (p *CBoxPage) buildCboxURL(section string) *url.URL {
}
func (p *CBoxPage) parsePage(document *goquery.Document) {
document.Find(".msg").Each(func(index int, element *goquery.Selection) {
messageIDString, exists := element.Attr("id")
if exists {
messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString))
if p.smallestID == -1 || messageIDInt < p.smallestID {
p.smallestID = messageIDInt
}
if p.largestID == -1 || messageIDInt > p.largestID {
p.largestID = messageIDInt
}
p.Messages[messageIDInt] = p.parseMessage(messageIDInt, element)
}
})
tableIndex := document.Find("table").First().Length()
// can paginate backwards on main page
_, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href")
if p.canPaginatePrevious {
cboxURL := p.buildCboxURL("archive")
q := cboxURL.Query()
q.Set("i", strconv.Itoa(p.smallestID))
p.previousString = q.Encode()
} else {
// in an archive page
previousString, canPrevious := document.Find("td[align='left'] a").Attr("href")
if canPrevious {
p.canPaginatePrevious = true
p.previousString = previousString[3:]
}
nextString, canNext := document.Find("td[align='right'] a").Attr("href")
if canNext {
p.canPaginateNext = true
p.nextString = nextString[3:]
}
}
}
func (p *CBoxPage) parseMessage(messageID int, element *goquery.Selection) *Message {
message := Message{
MessageID: messageID,
// tableIndex returns 0 when not found
if tableIndex > 0{
log.Println("Parsing old style page")
p.parseOldPage(document)
return
}
datetimeElement := element.Find("div").Text()
if datetimeElement != "" {
message.DateTime, _ = time.Parse(DatetimeFormat, datetimeElement)
}
element.Find("div").Remove()
name := element.Find("b.nme").Text()
if name != "" {
message.Username = name
}
element.Find("b.nme").Remove()
// trim the first 2 characters, colon and space
message.Message = element.Text()[2:]
return &message
p.isOld = false
log.Println("Parsing new style page")
p.parseNewPage(document)
}
func newHTTPRequest(url *url.URL) (*http.Request, error) {
+91
View File
@@ -0,0 +1,91 @@
package go_cbox_scraper
import (
"github.com/PuerkitoBio/goquery"
"strconv"
"time"
)
func (p *CBoxPage) parseNewPage(document *goquery.Document) {
document.Find(".msg").Each(func(index int, element *goquery.Selection) {
messageIDString, messageExists := element.Attr("data-id")
messageTime, timeExists := element.Attr("data-time")
if messageExists && timeExists {
messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString))
if p.smallestID == -1 || messageIDInt < p.smallestID {
p.smallestID = messageIDInt
}
if p.largestID == -1 || messageIDInt > p.largestID {
p.largestID = messageIDInt
}
intTime, err := strconv.ParseInt(messageTime, 10, 64)
if err != nil {
panic(err)
}
outMessageTime := time.Unix(intTime, 0).In(chicago)
p.Messages[messageIDInt] = p.parseNewMessage(messageIDInt, outMessageTime, element)
}
})
// can paginate backwards on main page
_, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href")
if p.canPaginatePrevious {
cboxURL := p.buildCboxURL("archive")
q := cboxURL.Query()
q.Set("i", strconv.Itoa(p.smallestID))
p.previousString = q.Encode()
}
}
func (p *CBoxPage) parseNewMessage(messageID int, timeIn time.Time, element *goquery.Selection) *Message {
message := Message{
MessageID: messageID,
DateTime: timeIn,
}
name := element.Find("div.nme").Text()
if name != "" {
message.Username = name
}
body := element.Find("div.body").Text()
if body != "" {
message.Message = body
}
return &message
}
func (p *CBoxPage) parseNewPageArchive(document *goquery.Document) {
document.Find(".msg").Each(func(index int, element *goquery.Selection) {
messageIDString, exists := element.Attr("id")
if exists {
messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString))
if p.smallestID == -1 || messageIDInt < p.smallestID {
p.smallestID = messageIDInt
}
if p.largestID == -1 || messageIDInt > p.largestID {
p.largestID = messageIDInt
}
p.Messages[messageIDInt] = p.parseOldMessage(messageIDInt, element)
}
})
// in an archive page
previousString, canPrevious := document.Find("td[align='left'] a").Attr("href")
p.canPaginatePrevious = canPrevious
if canPrevious {
p.previousString = previousString[3:]
}
nextString, canNext := document.Find("td[align='right'] a").Attr("href")
p.canPaginateNext = canNext
if canNext {
p.nextString = nextString[3:]
}
}
+68
View File
@@ -0,0 +1,68 @@
package go_cbox_scraper
import (
"github.com/PuerkitoBio/goquery"
"strconv"
"time"
)
func (p *CBoxPage) parseOldPage(document *goquery.Document) {
document.Find(".msg").Each(func(index int, element *goquery.Selection) {
messageIDString, exists := element.Attr("id")
if exists {
messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString))
if p.smallestID == -1 || messageIDInt < p.smallestID {
p.smallestID = messageIDInt
}
if p.largestID == -1 || messageIDInt > p.largestID {
p.largestID = messageIDInt
}
p.Messages[messageIDInt] = p.parseOldMessage(messageIDInt, element)
}
})
// can paginate backwards on main page
_, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href")
if p.canPaginatePrevious {
cboxURL := p.buildCboxURL("archive")
q := cboxURL.Query()
q.Set("i", strconv.Itoa(p.smallestID))
p.previousString = q.Encode()
} else {
// in an archive page
previousString, canPrevious := document.Find("td[align='left'] a").Attr("href")
if canPrevious {
p.canPaginatePrevious = true
p.previousString = previousString[3:]
}
nextString, canNext := document.Find("td[align='right'] a").Attr("href")
if canNext {
p.canPaginateNext = true
p.nextString = nextString[3:]
}
}
}
func (p *CBoxPage) parseOldMessage(messageID int, element *goquery.Selection) *Message {
message := Message{
MessageID: messageID,
}
datetimeElement := element.Find("div").Text()
if datetimeElement != "" {
message.DateTime, _ = time.Parse(DatetimeFormat, datetimeElement)
}
element.Find("div").Remove()
name := element.Find("b.nme").Text()
if name != "" {
message.Username = name
}
element.Find("b.nme").Remove()
// trim the first 2 characters, colon and space
message.Message = element.Text()[2:]
return &message
}
+1 -1
View File
@@ -6,7 +6,7 @@ import (
"time"
)
const DatetimeFormat = "2006-01-02 03:04PM"
const DatetimeFormat = "_2 Jan 06, 03:04 PM"
type Scraper struct {
SmallestMessageID int