update scraping logic to handle new style cbox pages and new unified dateformat
This commit is contained in:
1 parent
7ebc9d5acc
commit
a7d0094a08
5 files changed
+183
-57
No files matched your search
+3
-1
@@ -12,6 +12,8 @@ type Message struct {
|
||||
Message string
|
||||
}
|
||||
|
||||
const DisplayDatetimeFormat = "2006-01-02 03:04PM"
|
||||
|
||||
func (m *Message) String() string {
|
||||
return fmt.Sprintf("#%d [%s] <%s> %s", m.MessageID, m.DateTime.Format(DatetimeFormat), m.Username, m.Message)
|
||||
return fmt.Sprintf("#%d [%s] <%s> %s", m.MessageID, m.DateTime.Format(DisplayDatetimeFormat), m.Username, m.Message)
|
||||
}
|
||||
@@ -3,6 +3,7 @@ package go_cbox_scraper
|
||||
import (
|
||||
"errors"
|
||||
"fmt"
|
||||
"log"
|
||||
"net/http"
|
||||
"net/url"
|
||||
"regexp"
|
||||
@@ -14,10 +15,17 @@ import (
|
||||
|
||||
var (
|
||||
digitRegex *regexp.Regexp
|
||||
chicago *time.Location
|
||||
)
|
||||
|
||||
func init() {
|
||||
var err error
|
||||
|
||||
digitRegex = regexp.MustCompile(`\d+`)
|
||||
chicago, err = time.LoadLocation("America/Chicago")
|
||||
if err != nil {
|
||||
panic("Unable to load timezone offset for America/Chicago from local timezone database")
|
||||
}
|
||||
}
|
||||
|
||||
type CBoxPage struct {
|
||||
@@ -40,6 +48,8 @@ type CBoxPage struct {
|
||||
|
||||
section string
|
||||
|
||||
isOld bool
|
||||
|
||||
ServerInfo
|
||||
}
|
||||
|
||||
@@ -58,6 +68,7 @@ func NewCBoxPage(cbxInfo ServerInfo) *CBoxPage {
|
||||
smallestID: -1,
|
||||
largestID: -1,
|
||||
section: "",
|
||||
isOld: true,
|
||||
ServerInfo: cbxInfo,
|
||||
}
|
||||
}
|
||||
@@ -173,64 +184,18 @@ func (p *CBoxPage) buildCboxURL(section string) *url.URL {
|
||||
}
|
||||
|
||||
func (p *CBoxPage) parsePage(document *goquery.Document) {
|
||||
document.Find(".msg").Each(func(index int, element *goquery.Selection) {
|
||||
messageIDString, exists := element.Attr("id")
|
||||
if exists {
|
||||
messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString))
|
||||
if p.smallestID == -1 || messageIDInt < p.smallestID {
|
||||
p.smallestID = messageIDInt
|
||||
}
|
||||
if p.largestID == -1 || messageIDInt > p.largestID {
|
||||
p.largestID = messageIDInt
|
||||
}
|
||||
p.Messages[messageIDInt] = p.parseMessage(messageIDInt, element)
|
||||
}
|
||||
})
|
||||
tableIndex := document.Find("table").First().Length()
|
||||
|
||||
// can paginate backwards on main page
|
||||
_, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href")
|
||||
if p.canPaginatePrevious {
|
||||
cboxURL := p.buildCboxURL("archive")
|
||||
q := cboxURL.Query()
|
||||
q.Set("i", strconv.Itoa(p.smallestID))
|
||||
p.previousString = q.Encode()
|
||||
} else {
|
||||
// in an archive page
|
||||
previousString, canPrevious := document.Find("td[align='left'] a").Attr("href")
|
||||
if canPrevious {
|
||||
p.canPaginatePrevious = true
|
||||
p.previousString = previousString[3:]
|
||||
}
|
||||
|
||||
nextString, canNext := document.Find("td[align='right'] a").Attr("href")
|
||||
if canNext {
|
||||
p.canPaginateNext = true
|
||||
p.nextString = nextString[3:]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (p *CBoxPage) parseMessage(messageID int, element *goquery.Selection) *Message {
|
||||
message := Message{
|
||||
MessageID: messageID,
|
||||
// tableIndex returns 0 when not found
|
||||
if tableIndex > 0{
|
||||
log.Println("Parsing old style page")
|
||||
p.parseOldPage(document)
|
||||
return
|
||||
}
|
||||
|
||||
datetimeElement := element.Find("div").Text()
|
||||
if datetimeElement != "" {
|
||||
message.DateTime, _ = time.Parse(DatetimeFormat, datetimeElement)
|
||||
}
|
||||
element.Find("div").Remove()
|
||||
|
||||
name := element.Find("b.nme").Text()
|
||||
if name != "" {
|
||||
message.Username = name
|
||||
}
|
||||
element.Find("b.nme").Remove()
|
||||
|
||||
// trim the first 2 characters, colon and space
|
||||
message.Message = element.Text()[2:]
|
||||
|
||||
return &message
|
||||
p.isOld = false
|
||||
log.Println("Parsing new style page")
|
||||
p.parseNewPage(document)
|
||||
}
|
||||
|
||||
func newHTTPRequest(url *url.URL) (*http.Request, error) {
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
package go_cbox_scraper
|
||||
|
||||
import (
|
||||
"github.com/PuerkitoBio/goquery"
|
||||
"strconv"
|
||||
"time"
|
||||
)
|
||||
|
||||
func (p *CBoxPage) parseNewPage(document *goquery.Document) {
|
||||
document.Find(".msg").Each(func(index int, element *goquery.Selection) {
|
||||
messageIDString, messageExists := element.Attr("data-id")
|
||||
messageTime, timeExists := element.Attr("data-time")
|
||||
if messageExists && timeExists {
|
||||
messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString))
|
||||
if p.smallestID == -1 || messageIDInt < p.smallestID {
|
||||
p.smallestID = messageIDInt
|
||||
}
|
||||
if p.largestID == -1 || messageIDInt > p.largestID {
|
||||
p.largestID = messageIDInt
|
||||
}
|
||||
|
||||
intTime, err := strconv.ParseInt(messageTime, 10, 64)
|
||||
if err != nil {
|
||||
panic(err)
|
||||
}
|
||||
|
||||
outMessageTime := time.Unix(intTime, 0).In(chicago)
|
||||
|
||||
p.Messages[messageIDInt] = p.parseNewMessage(messageIDInt, outMessageTime, element)
|
||||
}
|
||||
})
|
||||
|
||||
// can paginate backwards on main page
|
||||
_, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href")
|
||||
if p.canPaginatePrevious {
|
||||
cboxURL := p.buildCboxURL("archive")
|
||||
q := cboxURL.Query()
|
||||
q.Set("i", strconv.Itoa(p.smallestID))
|
||||
p.previousString = q.Encode()
|
||||
}
|
||||
|
||||
}
|
||||
|
||||
func (p *CBoxPage) parseNewMessage(messageID int, timeIn time.Time, element *goquery.Selection) *Message {
|
||||
message := Message{
|
||||
MessageID: messageID,
|
||||
DateTime: timeIn,
|
||||
}
|
||||
|
||||
name := element.Find("div.nme").Text()
|
||||
if name != "" {
|
||||
message.Username = name
|
||||
}
|
||||
|
||||
body := element.Find("div.body").Text()
|
||||
if body != "" {
|
||||
message.Message = body
|
||||
}
|
||||
|
||||
return &message
|
||||
}
|
||||
|
||||
func (p *CBoxPage) parseNewPageArchive(document *goquery.Document) {
|
||||
document.Find(".msg").Each(func(index int, element *goquery.Selection) {
|
||||
messageIDString, exists := element.Attr("id")
|
||||
if exists {
|
||||
messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString))
|
||||
if p.smallestID == -1 || messageIDInt < p.smallestID {
|
||||
p.smallestID = messageIDInt
|
||||
}
|
||||
if p.largestID == -1 || messageIDInt > p.largestID {
|
||||
p.largestID = messageIDInt
|
||||
}
|
||||
p.Messages[messageIDInt] = p.parseOldMessage(messageIDInt, element)
|
||||
}
|
||||
})
|
||||
|
||||
// in an archive page
|
||||
previousString, canPrevious := document.Find("td[align='left'] a").Attr("href")
|
||||
p.canPaginatePrevious = canPrevious
|
||||
if canPrevious {
|
||||
p.previousString = previousString[3:]
|
||||
}
|
||||
|
||||
nextString, canNext := document.Find("td[align='right'] a").Attr("href")
|
||||
p.canPaginateNext = canNext
|
||||
if canNext {
|
||||
p.nextString = nextString[3:]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,68 @@
|
||||
package go_cbox_scraper
|
||||
|
||||
import (
|
||||
"github.com/PuerkitoBio/goquery"
|
||||
"strconv"
|
||||
"time"
|
||||
)
|
||||
|
||||
func (p *CBoxPage) parseOldPage(document *goquery.Document) {
|
||||
document.Find(".msg").Each(func(index int, element *goquery.Selection) {
|
||||
messageIDString, exists := element.Attr("id")
|
||||
if exists {
|
||||
messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString))
|
||||
if p.smallestID == -1 || messageIDInt < p.smallestID {
|
||||
p.smallestID = messageIDInt
|
||||
}
|
||||
if p.largestID == -1 || messageIDInt > p.largestID {
|
||||
p.largestID = messageIDInt
|
||||
}
|
||||
p.Messages[messageIDInt] = p.parseOldMessage(messageIDInt, element)
|
||||
}
|
||||
})
|
||||
|
||||
// can paginate backwards on main page
|
||||
_, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href")
|
||||
if p.canPaginatePrevious {
|
||||
cboxURL := p.buildCboxURL("archive")
|
||||
q := cboxURL.Query()
|
||||
q.Set("i", strconv.Itoa(p.smallestID))
|
||||
p.previousString = q.Encode()
|
||||
} else {
|
||||
// in an archive page
|
||||
previousString, canPrevious := document.Find("td[align='left'] a").Attr("href")
|
||||
if canPrevious {
|
||||
p.canPaginatePrevious = true
|
||||
p.previousString = previousString[3:]
|
||||
}
|
||||
|
||||
nextString, canNext := document.Find("td[align='right'] a").Attr("href")
|
||||
if canNext {
|
||||
p.canPaginateNext = true
|
||||
p.nextString = nextString[3:]
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func (p *CBoxPage) parseOldMessage(messageID int, element *goquery.Selection) *Message {
|
||||
message := Message{
|
||||
MessageID: messageID,
|
||||
}
|
||||
|
||||
datetimeElement := element.Find("div").Text()
|
||||
if datetimeElement != "" {
|
||||
message.DateTime, _ = time.Parse(DatetimeFormat, datetimeElement)
|
||||
}
|
||||
element.Find("div").Remove()
|
||||
|
||||
name := element.Find("b.nme").Text()
|
||||
if name != "" {
|
||||
message.Username = name
|
||||
}
|
||||
element.Find("b.nme").Remove()
|
||||
|
||||
// trim the first 2 characters, colon and space
|
||||
message.Message = element.Text()[2:]
|
||||
|
||||
return &message
|
||||
}
|
||||
+1
-1
@@ -6,7 +6,7 @@ import (
|
||||
"time"
|
||||
)
|
||||
|
||||
const DatetimeFormat = "2006-01-02 03:04PM"
|
||||
const DatetimeFormat = "_2 Jan 06, 03:04 PM"
|
||||
|
||||
type Scraper struct {
|
||||
SmallestMessageID int
|
||||
|
||||
Reference in new issue
Block a user