From a7d0094a0874beebbf8b9beb175c7e48f339ca9a Mon Sep 17 00:00:00 2001 From: Neo-Desktop Date: Mon, 4 Oct 2021 09:00:49 -0700 Subject: [PATCH] update scraping logic to handle new style cbox pages and new unified dateformat --- message.go | 4 ++- page.go | 75 +++++++++++---------------------------- page_newstyle.go | 91 ++++++++++++++++++++++++++++++++++++++++++++++++ page_oldstyle.go | 68 ++++++++++++++++++++++++++++++++++++ scraper.go | 2 +- 5 files changed, 183 insertions(+), 57 deletions(-) create mode 100644 page_newstyle.go create mode 100644 page_oldstyle.go diff --git a/message.go b/message.go index d33cd4c..bc7228f 100644 --- a/message.go +++ b/message.go @@ -12,6 +12,8 @@ type Message struct { Message string } +const DisplayDatetimeFormat = "2006-01-02 03:04PM" + func (m *Message) String() string { - return fmt.Sprintf("#%d [%s] <%s> %s", m.MessageID, m.DateTime.Format(DatetimeFormat), m.Username, m.Message) + return fmt.Sprintf("#%d [%s] <%s> %s", m.MessageID, m.DateTime.Format(DisplayDatetimeFormat), m.Username, m.Message) } diff --git a/page.go b/page.go index 82db1ea..16c14b1 100644 --- a/page.go +++ b/page.go @@ -3,6 +3,7 @@ package go_cbox_scraper import ( "errors" "fmt" + "log" "net/http" "net/url" "regexp" @@ -14,10 +15,17 @@ import ( var ( digitRegex *regexp.Regexp + chicago *time.Location ) func init() { + var err error + digitRegex = regexp.MustCompile(`\d+`) + chicago, err = time.LoadLocation("America/Chicago") + if err != nil { + panic("Unable to load timezone offset for America/Chicago from local timezone database") + } } type CBoxPage struct { @@ -40,6 +48,8 @@ type CBoxPage struct { section string + isOld bool + ServerInfo } @@ -58,6 +68,7 @@ func NewCBoxPage(cbxInfo ServerInfo) *CBoxPage { smallestID: -1, largestID: -1, section: "", + isOld: true, ServerInfo: cbxInfo, } } @@ -173,64 +184,18 @@ func (p *CBoxPage) buildCboxURL(section string) *url.URL { } func (p *CBoxPage) parsePage(document *goquery.Document) { - document.Find(".msg").Each(func(index int, element *goquery.Selection) { - messageIDString, exists := element.Attr("id") - if exists { - messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString)) - if p.smallestID == -1 || messageIDInt < p.smallestID { - p.smallestID = messageIDInt - } - if p.largestID == -1 || messageIDInt > p.largestID { - p.largestID = messageIDInt - } - p.Messages[messageIDInt] = p.parseMessage(messageIDInt, element) - } - }) + tableIndex := document.Find("table").First().Length() - // can paginate backwards on main page - _, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href") - if p.canPaginatePrevious { - cboxURL := p.buildCboxURL("archive") - q := cboxURL.Query() - q.Set("i", strconv.Itoa(p.smallestID)) - p.previousString = q.Encode() - } else { - // in an archive page - previousString, canPrevious := document.Find("td[align='left'] a").Attr("href") - if canPrevious { - p.canPaginatePrevious = true - p.previousString = previousString[3:] - } - - nextString, canNext := document.Find("td[align='right'] a").Attr("href") - if canNext { - p.canPaginateNext = true - p.nextString = nextString[3:] - } - } -} - -func (p *CBoxPage) parseMessage(messageID int, element *goquery.Selection) *Message { - message := Message{ - MessageID: messageID, + // tableIndex returns 0 when not found + if tableIndex > 0{ + log.Println("Parsing old style page") + p.parseOldPage(document) + return } - datetimeElement := element.Find("div").Text() - if datetimeElement != "" { - message.DateTime, _ = time.Parse(DatetimeFormat, datetimeElement) - } - element.Find("div").Remove() - - name := element.Find("b.nme").Text() - if name != "" { - message.Username = name - } - element.Find("b.nme").Remove() - - // trim the first 2 characters, colon and space - message.Message = element.Text()[2:] - - return &message + p.isOld = false + log.Println("Parsing new style page") + p.parseNewPage(document) } func newHTTPRequest(url *url.URL) (*http.Request, error) { diff --git a/page_newstyle.go b/page_newstyle.go new file mode 100644 index 0000000..3a3afe4 --- /dev/null +++ b/page_newstyle.go @@ -0,0 +1,91 @@ +package go_cbox_scraper + +import ( + "github.com/PuerkitoBio/goquery" + "strconv" + "time" +) + +func (p *CBoxPage) parseNewPage(document *goquery.Document) { + document.Find(".msg").Each(func(index int, element *goquery.Selection) { + messageIDString, messageExists := element.Attr("data-id") + messageTime, timeExists := element.Attr("data-time") + if messageExists && timeExists { + messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString)) + if p.smallestID == -1 || messageIDInt < p.smallestID { + p.smallestID = messageIDInt + } + if p.largestID == -1 || messageIDInt > p.largestID { + p.largestID = messageIDInt + } + + intTime, err := strconv.ParseInt(messageTime, 10, 64) + if err != nil { + panic(err) + } + + outMessageTime := time.Unix(intTime, 0).In(chicago) + + p.Messages[messageIDInt] = p.parseNewMessage(messageIDInt, outMessageTime, element) + } + }) + + // can paginate backwards on main page + _, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href") + if p.canPaginatePrevious { + cboxURL := p.buildCboxURL("archive") + q := cboxURL.Query() + q.Set("i", strconv.Itoa(p.smallestID)) + p.previousString = q.Encode() + } + +} + +func (p *CBoxPage) parseNewMessage(messageID int, timeIn time.Time, element *goquery.Selection) *Message { + message := Message{ + MessageID: messageID, + DateTime: timeIn, + } + + name := element.Find("div.nme").Text() + if name != "" { + message.Username = name + } + + body := element.Find("div.body").Text() + if body != "" { + message.Message = body + } + + return &message +} + +func (p *CBoxPage) parseNewPageArchive(document *goquery.Document) { + document.Find(".msg").Each(func(index int, element *goquery.Selection) { + messageIDString, exists := element.Attr("id") + if exists { + messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString)) + if p.smallestID == -1 || messageIDInt < p.smallestID { + p.smallestID = messageIDInt + } + if p.largestID == -1 || messageIDInt > p.largestID { + p.largestID = messageIDInt + } + p.Messages[messageIDInt] = p.parseOldMessage(messageIDInt, element) + } + }) + + // in an archive page + previousString, canPrevious := document.Find("td[align='left'] a").Attr("href") + p.canPaginatePrevious = canPrevious + if canPrevious { + p.previousString = previousString[3:] + } + + nextString, canNext := document.Find("td[align='right'] a").Attr("href") + p.canPaginateNext = canNext + if canNext { + p.nextString = nextString[3:] + } +} + diff --git a/page_oldstyle.go b/page_oldstyle.go new file mode 100644 index 0000000..83057cd --- /dev/null +++ b/page_oldstyle.go @@ -0,0 +1,68 @@ +package go_cbox_scraper + +import ( + "github.com/PuerkitoBio/goquery" + "strconv" + "time" +) + +func (p *CBoxPage) parseOldPage(document *goquery.Document) { + document.Find(".msg").Each(func(index int, element *goquery.Selection) { + messageIDString, exists := element.Attr("id") + if exists { + messageIDInt, _ := strconv.Atoi(digitRegex.FindString(messageIDString)) + if p.smallestID == -1 || messageIDInt < p.smallestID { + p.smallestID = messageIDInt + } + if p.largestID == -1 || messageIDInt > p.largestID { + p.largestID = messageIDInt + } + p.Messages[messageIDInt] = p.parseOldMessage(messageIDInt, element) + } + }) + + // can paginate backwards on main page + _, p.canPaginatePrevious = document.Find("#lnkArchive").Attr("href") + if p.canPaginatePrevious { + cboxURL := p.buildCboxURL("archive") + q := cboxURL.Query() + q.Set("i", strconv.Itoa(p.smallestID)) + p.previousString = q.Encode() + } else { + // in an archive page + previousString, canPrevious := document.Find("td[align='left'] a").Attr("href") + if canPrevious { + p.canPaginatePrevious = true + p.previousString = previousString[3:] + } + + nextString, canNext := document.Find("td[align='right'] a").Attr("href") + if canNext { + p.canPaginateNext = true + p.nextString = nextString[3:] + } + } +} + +func (p *CBoxPage) parseOldMessage(messageID int, element *goquery.Selection) *Message { + message := Message{ + MessageID: messageID, + } + + datetimeElement := element.Find("div").Text() + if datetimeElement != "" { + message.DateTime, _ = time.Parse(DatetimeFormat, datetimeElement) + } + element.Find("div").Remove() + + name := element.Find("b.nme").Text() + if name != "" { + message.Username = name + } + element.Find("b.nme").Remove() + + // trim the first 2 characters, colon and space + message.Message = element.Text()[2:] + + return &message +} \ No newline at end of file diff --git a/scraper.go b/scraper.go index e312a82..cb5ac12 100644 --- a/scraper.go +++ b/scraper.go @@ -6,7 +6,7 @@ import ( "time" ) -const DatetimeFormat = "2006-01-02 03:04PM" +const DatetimeFormat = "_2 Jan 06, 03:04 PM" type Scraper struct { SmallestMessageID int