From 2f96d51ac363cb35e2ae05df363265daefa6a9cf Mon Sep 17 00:00:00 2001 From: "Matthew R. Wilson" Date: Fri, 14 Nov 2025 23:54:11 -0800 Subject: [PATCH] Groundwork for multi codepage support --- ebcdic.go | 169 +++++++++++++++++++++++++++++++++++----------------- response.go | 4 +- screen.go | 2 +- 3 files changed, 116 insertions(+), 59 deletions(-) diff --git a/ebcdic.go b/ebcdic.go index 3ee1727..d737ecb 100644 --- a/ebcdic.go +++ b/ebcdic.go @@ -8,8 +8,18 @@ import ( "unicode/utf8" ) -// After careful consideration, I have decided that the code page we will -// support for EBCDIC is IBM CP 1047. +// Implementations of Charset provide EBCDIC<->UTF-8 translation. +type Charset interface { + // Decode converts a slice of EBCDIC bytes into a UTF-8 string. + Decode(e []byte) string + + // Encode converts a UTF-8 string into a slice of EBCDIC bytes. + Encode(s string) []byte +} + +// After careful consideration, I have decided that the default code page we +// will support for EBCDIC is IBM CP 1047. Other code pages may be globally +// selected with the SetCodepage() function. // // In suite3270 (e.g. c3270/x3270), the default code page is what it calls // "brackets". This is CP37 with the [, ], Ý, and ¨ characters swapped around. @@ -28,54 +38,101 @@ import ( // In IBM PCOMM, CP37 is the default. For correct mapping of [, ], Ý, ¨, ^, // and ¬, you must switch the session parameters from "037 United States" to // "1047 United States". +var currentCodepage Charset = Codepage1047() + +// SetCodepage sets the codepage/character set that go3270 uses. This is a +// global setting, so if you're expecting clients to be configured to use a +// character set other than go3270's default, cp1047, you should probably set +// this during your application initialization and then leave it unchanged +// after. This is _not_ a per-connection setting. +func SetCodepage(cs Charset) { + currentCodepage = cs +} + +// Internal implementation of the Charset interface we'll use for the codepage +// support we provide. +type charset struct { + // EBCDIC byte to Unicode code point for bytes 0x00-0xFF + e2u []rune + + // Unicode code point to EBCDIC byte for codepoints 0x00-0xFF + u2e []byte + + // Map of Unicode code points to EBCDIC bytes for codepoints >0xFF + highu2e map[rune]byte + + // The EBCDIC substitute character to use if there is no EBCDIC character + // for the requested Unicode code point (typically 0x3F). + esub byte + + // The "graphic escape" EBCDIC byte (is it ever anything other than 0x0E?) + ge byte + + // Graphic escape codepage EBCDIC byte to Unicode code point for bytes + // 0x00-0xFF. Use rune '�' for unmapped bytes. + ge2u []rune + + // Map of Unicode code points to graphic escape EBCDIC bytes. + u2ge map[rune]byte +} // IBM CP 1047 <-> Unicode mappings from: // https://raw.githubusercontent.com/unicode-org/icu-data/refs/heads/main/charset/data/ucm/glibc-IBM1047-2.1.2.ucm - -var cp1047ToUnicode = []rune{ - /* x0 x1 x2 x3 x4 x5 x6 x7 x8 x9 xA xB xC xD xE xF */ - /* 0x */ 0x00, 0x01, 0x02, 0x03, 0x9C, 0x09, 0x86, 0x7F, 0x97, 0x8D, 0x8E, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, - /* 1x */ 0x10, 0x11, 0x12, 0x13, 0x9D, 0x85, 0x08, 0x87, 0x18, 0x19, 0x92, 0x8F, 0x1C, 0x1D, 0x1E, 0x1F, - /* 2x */ 0x80, 0x81, 0x82, 0x83, 0x84, 0x0A, 0x17, 0x1B, 0x88, 0x89, 0x8A, 0x8B, 0x8C, 0x05, 0x06, 0x07, - /* 3x */ 0x90, 0x91, 0x16, 0x93, 0x94, 0x95, 0x96, 0x04, 0x98, 0x99, 0x9A, 0x9B, 0x14, 0x15, 0x9E, 0x1A, - /* 4x */ 0x20, 0xA0, 0xE2, 0xE4, 0xE0, 0xE1, 0xE3, 0xE5, 0xE7, 0xF1, 0xA2, 0x2E, 0x3C, 0x28, 0x2B, 0x7C, - /* 5x */ 0x26, 0xE9, 0xEA, 0xEB, 0xE8, 0xED, 0xEE, 0xEF, 0xEC, 0xDF, 0x21, 0x24, 0x2A, 0x29, 0x3B, 0x5E, - /* 6x */ 0x2D, 0x2F, 0xC2, 0xC4, 0xC0, 0xC1, 0xC3, 0xC5, 0xC7, 0xD1, 0xA6, 0x2C, 0x25, 0x5F, 0x3E, 0x3F, - /* 7x */ 0xF8, 0xC9, 0xCA, 0xCB, 0xC8, 0xCD, 0xCE, 0xCF, 0xCC, 0x60, 0x3A, 0x23, 0x40, 0x27, 0x3D, 0x22, - /* 8x */ 0xD8, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, 0x67, 0x68, 0x69, 0xAB, 0xBB, 0xF0, 0xFD, 0xFE, 0xB1, - /* 9x */ 0xB0, 0x6A, 0x6B, 0x6C, 0x6D, 0x6E, 0x6F, 0x70, 0x71, 0x72, 0xAA, 0xBA, 0xE6, 0xB8, 0xC6, 0xA4, - /* Ax */ 0xB5, 0x7E, 0x73, 0x74, 0x75, 0x76, 0x77, 0x78, 0x79, 0x7A, 0xA1, 0xBF, 0xD0, 0x5B, 0xDE, 0xAE, - /* Bx */ 0xAC, 0xA3, 0xA5, 0xB7, 0xA9, 0xA7, 0xB6, 0xBC, 0xBD, 0xBE, 0xDD, 0xA8, 0xAF, 0x5D, 0xB4, 0xD7, - /* Cx */ 0x7B, 0x41, 0x42, 0x43, 0x44, 0x45, 0x46, 0x47, 0x48, 0x49, 0xAD, 0xF4, 0xF6, 0xF2, 0xF3, 0xF5, - /* Dx */ 0x7D, 0x4A, 0x4B, 0x4C, 0x4D, 0x4E, 0x4F, 0x50, 0x51, 0x52, 0xB9, 0xFB, 0xFC, 0xF9, 0xFA, 0xFF, - /* Ex */ 0x5C, 0xF7, 0x53, 0x54, 0x55, 0x56, 0x57, 0x58, 0x59, 0x5A, 0xB2, 0xD4, 0xD6, 0xD2, 0xD3, 0xD5, - /* Fx */ 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, 0x38, 0x39, 0xB3, 0xDB, 0xDC, 0xD9, 0xDA, 0x9F, +func Codepage1047() Charset { + return &charset{ + e2u: []rune{ + /* x0 x1 x2 x3 x4 x5 x6 x7 x8 x9 xA xB xC xD xE xF */ + /* 0x */ 0x00, 0x01, 0x02, 0x03, 0x9C, 0x09, 0x86, 0x7F, 0x97, 0x8D, 0x8E, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, + /* 1x */ 0x10, 0x11, 0x12, 0x13, 0x9D, 0x85, 0x08, 0x87, 0x18, 0x19, 0x92, 0x8F, 0x1C, 0x1D, 0x1E, 0x1F, + /* 2x */ 0x80, 0x81, 0x82, 0x83, 0x84, 0x0A, 0x17, 0x1B, 0x88, 0x89, 0x8A, 0x8B, 0x8C, 0x05, 0x06, 0x07, + /* 3x */ 0x90, 0x91, 0x16, 0x93, 0x94, 0x95, 0x96, 0x04, 0x98, 0x99, 0x9A, 0x9B, 0x14, 0x15, 0x9E, 0x1A, + /* 4x */ 0x20, 0xA0, 0xE2, 0xE4, 0xE0, 0xE1, 0xE3, 0xE5, 0xE7, 0xF1, 0xA2, 0x2E, 0x3C, 0x28, 0x2B, 0x7C, + /* 5x */ 0x26, 0xE9, 0xEA, 0xEB, 0xE8, 0xED, 0xEE, 0xEF, 0xEC, 0xDF, 0x21, 0x24, 0x2A, 0x29, 0x3B, 0x5E, + /* 6x */ 0x2D, 0x2F, 0xC2, 0xC4, 0xC0, 0xC1, 0xC3, 0xC5, 0xC7, 0xD1, 0xA6, 0x2C, 0x25, 0x5F, 0x3E, 0x3F, + /* 7x */ 0xF8, 0xC9, 0xCA, 0xCB, 0xC8, 0xCD, 0xCE, 0xCF, 0xCC, 0x60, 0x3A, 0x23, 0x40, 0x27, 0x3D, 0x22, + /* 8x */ 0xD8, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, 0x67, 0x68, 0x69, 0xAB, 0xBB, 0xF0, 0xFD, 0xFE, 0xB1, + /* 9x */ 0xB0, 0x6A, 0x6B, 0x6C, 0x6D, 0x6E, 0x6F, 0x70, 0x71, 0x72, 0xAA, 0xBA, 0xE6, 0xB8, 0xC6, 0xA4, + /* Ax */ 0xB5, 0x7E, 0x73, 0x74, 0x75, 0x76, 0x77, 0x78, 0x79, 0x7A, 0xA1, 0xBF, 0xD0, 0x5B, 0xDE, 0xAE, + /* Bx */ 0xAC, 0xA3, 0xA5, 0xB7, 0xA9, 0xA7, 0xB6, 0xBC, 0xBD, 0xBE, 0xDD, 0xA8, 0xAF, 0x5D, 0xB4, 0xD7, + /* Cx */ 0x7B, 0x41, 0x42, 0x43, 0x44, 0x45, 0x46, 0x47, 0x48, 0x49, 0xAD, 0xF4, 0xF6, 0xF2, 0xF3, 0xF5, + /* Dx */ 0x7D, 0x4A, 0x4B, 0x4C, 0x4D, 0x4E, 0x4F, 0x50, 0x51, 0x52, 0xB9, 0xFB, 0xFC, 0xF9, 0xFA, 0xFF, + /* Ex */ 0x5C, 0xF7, 0x53, 0x54, 0x55, 0x56, 0x57, 0x58, 0x59, 0x5A, 0xB2, 0xD4, 0xD6, 0xD2, 0xD3, 0xD5, + /* Fx */ 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, 0x38, 0x39, 0xB3, 0xDB, 0xDC, 0xD9, 0xDA, 0x9F, + }, + u2e: []byte{ + /* x0 x1 x2 x3 x4 x5 x6 x7 x8 x9 xA xB xC xD xE xF */ + /* 0x */ 0x00, 0x01, 0x02, 0x03, 0x37, 0x2D, 0x2E, 0x2F, 0x16, 0x05, 0x25, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, + /* 1x */ 0x10, 0x11, 0x12, 0x13, 0x3C, 0x3D, 0x32, 0x26, 0x18, 0x19, 0x3F, 0x27, 0x1C, 0x1D, 0x1E, 0x1F, + /* 2x */ 0x40, 0x5A, 0x7F, 0x7B, 0x5B, 0x6C, 0x50, 0x7D, 0x4D, 0x5D, 0x5C, 0x4E, 0x6B, 0x60, 0x4B, 0x61, + /* 3x */ 0xF0, 0xF1, 0xF2, 0xF3, 0xF4, 0xF5, 0xF6, 0xF7, 0xF8, 0xF9, 0x7A, 0x5E, 0x4C, 0x7E, 0x6E, 0x6F, + /* 4x */ 0x7C, 0xC1, 0xC2, 0xC3, 0xC4, 0xC5, 0xC6, 0xC7, 0xC8, 0xC9, 0xD1, 0xD2, 0xD3, 0xD4, 0xD5, 0xD6, + /* 5x */ 0xD7, 0xD8, 0xD9, 0xE2, 0xE3, 0xE4, 0xE5, 0xE6, 0xE7, 0xE8, 0xE9, 0xAD, 0xE0, 0xBD, 0x5F, 0x6D, + /* 6x */ 0x79, 0x81, 0x82, 0x83, 0x84, 0x85, 0x86, 0x87, 0x88, 0x89, 0x91, 0x92, 0x93, 0x94, 0x95, 0x96, + /* 7x */ 0x97, 0x98, 0x99, 0xA2, 0xA3, 0xA4, 0xA5, 0xA6, 0xA7, 0xA8, 0xA9, 0xC0, 0x4F, 0xD0, 0xA1, 0x07, + /* 8x */ 0x20, 0x21, 0x22, 0x23, 0x24, 0x15, 0x06, 0x17, 0x28, 0x29, 0x2A, 0x2B, 0x2C, 0x09, 0x0A, 0x1B, + /* 9x */ 0x30, 0x31, 0x1A, 0x33, 0x34, 0x35, 0x36, 0x08, 0x38, 0x39, 0x3A, 0x3B, 0x04, 0x14, 0x3E, 0xFF, + /* Ax */ 0x41, 0xAA, 0x4A, 0xB1, 0x9F, 0xB2, 0x6A, 0xB5, 0xBB, 0xB4, 0x9A, 0x8A, 0xB0, 0xCA, 0xAF, 0xBC, + /* Bx */ 0x90, 0x8F, 0xEA, 0xFA, 0xBE, 0xA0, 0xB6, 0xB3, 0x9D, 0xDA, 0x9B, 0x8B, 0xB7, 0xB8, 0xB9, 0xAB, + /* Cx */ 0x64, 0x65, 0x62, 0x66, 0x63, 0x67, 0x9E, 0x68, 0x74, 0x71, 0x72, 0x73, 0x78, 0x75, 0x76, 0x77, + /* Dx */ 0xAC, 0x69, 0xED, 0xEE, 0xEB, 0xEF, 0xEC, 0xBF, 0x80, 0xFD, 0xFE, 0xFB, 0xFC, 0xBA, 0xAE, 0x59, + /* Ex */ 0x44, 0x45, 0x42, 0x46, 0x43, 0x47, 0x9C, 0x48, 0x54, 0x51, 0x52, 0x53, 0x58, 0x55, 0x56, 0x57, + /* Fx */ 0x8C, 0x49, 0xCD, 0xCE, 0xCB, 0xCF, 0xCC, 0xE1, 0x70, 0xDD, 0xDE, 0xDB, 0xDC, 0x8D, 0x8E, 0xDF, + }, + highu2e: map[rune]byte{}, + esub: 0x3f, + ge: 0x08, + ge2u: cp310ToUnicode, + u2ge: unicodeToCP310, + } } -var unicodeToCP1047 = []byte{ - /* x0 x1 x2 x3 x4 x5 x6 x7 x8 x9 xA xB xC xD xE xF */ - /* 0x */ 0x00, 0x01, 0x02, 0x03, 0x37, 0x2D, 0x2E, 0x2F, 0x16, 0x05, 0x25, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F, - /* 1x */ 0x10, 0x11, 0x12, 0x13, 0x3C, 0x3D, 0x32, 0x26, 0x18, 0x19, 0x3F, 0x27, 0x1C, 0x1D, 0x1E, 0x1F, - /* 2x */ 0x40, 0x5A, 0x7F, 0x7B, 0x5B, 0x6C, 0x50, 0x7D, 0x4D, 0x5D, 0x5C, 0x4E, 0x6B, 0x60, 0x4B, 0x61, - /* 3x */ 0xF0, 0xF1, 0xF2, 0xF3, 0xF4, 0xF5, 0xF6, 0xF7, 0xF8, 0xF9, 0x7A, 0x5E, 0x4C, 0x7E, 0x6E, 0x6F, - /* 4x */ 0x7C, 0xC1, 0xC2, 0xC3, 0xC4, 0xC5, 0xC6, 0xC7, 0xC8, 0xC9, 0xD1, 0xD2, 0xD3, 0xD4, 0xD5, 0xD6, - /* 5x */ 0xD7, 0xD8, 0xD9, 0xE2, 0xE3, 0xE4, 0xE5, 0xE6, 0xE7, 0xE8, 0xE9, 0xAD, 0xE0, 0xBD, 0x5F, 0x6D, - /* 6x */ 0x79, 0x81, 0x82, 0x83, 0x84, 0x85, 0x86, 0x87, 0x88, 0x89, 0x91, 0x92, 0x93, 0x94, 0x95, 0x96, - /* 7x */ 0x97, 0x98, 0x99, 0xA2, 0xA3, 0xA4, 0xA5, 0xA6, 0xA7, 0xA8, 0xA9, 0xC0, 0x4F, 0xD0, 0xA1, 0x07, - /* 8x */ 0x20, 0x21, 0x22, 0x23, 0x24, 0x15, 0x06, 0x17, 0x28, 0x29, 0x2A, 0x2B, 0x2C, 0x09, 0x0A, 0x1B, - /* 9x */ 0x30, 0x31, 0x1A, 0x33, 0x34, 0x35, 0x36, 0x08, 0x38, 0x39, 0x3A, 0x3B, 0x04, 0x14, 0x3E, 0xFF, - /* Ax */ 0x41, 0xAA, 0x4A, 0xB1, 0x9F, 0xB2, 0x6A, 0xB5, 0xBB, 0xB4, 0x9A, 0x8A, 0xB0, 0xCA, 0xAF, 0xBC, - /* Bx */ 0x90, 0x8F, 0xEA, 0xFA, 0xBE, 0xA0, 0xB6, 0xB3, 0x9D, 0xDA, 0x9B, 0x8B, 0xB7, 0xB8, 0xB9, 0xAB, - /* Cx */ 0x64, 0x65, 0x62, 0x66, 0x63, 0x67, 0x9E, 0x68, 0x74, 0x71, 0x72, 0x73, 0x78, 0x75, 0x76, 0x77, - /* Dx */ 0xAC, 0x69, 0xED, 0xEE, 0xEB, 0xEF, 0xEC, 0xBF, 0x80, 0xFD, 0xFE, 0xFB, 0xFC, 0xBA, 0xAE, 0x59, - /* Ex */ 0x44, 0x45, 0x42, 0x46, 0x43, 0x47, 0x9C, 0x48, 0x54, 0x51, 0x52, 0x53, 0x58, 0x55, 0x56, 0x57, - /* Fx */ 0x8C, 0x49, 0xCD, 0xCE, 0xCB, 0xCF, 0xCC, 0xE1, 0x70, 0xDD, 0xDE, 0xDB, 0xDC, 0x8D, 0x8E, 0xDF, -} - -// Furthermore, certain characters are supported in the "graphic escape" -// CP310. These are arbitrary Unicode code points, so we will look them up via -// a map. For simplicity of our mapping implementation, we will not support the -// italic underlined A-Z characters that require combining characters. +// Certain characters are supported in the "graphic escape" CP310. These are +// arbitrary Unicode code points, so we will look them up via a map. For +// simplicity of our mapping implementation, we will not support the italic +// underlined A-Z characters that require combining characters. +// +// We will share this map among all of the codepages that we provide +// implementations for. // // https://public.dhe.ibm.com/software/globalization/gcoc/attachments/CP00310.pdf var unicodeToCP310 = map[rune]byte{ @@ -127,25 +184,25 @@ var cp310ToUnicode = []rune{ // decode will convert a CP1047 byte array into a UTF-8 Go string, handling // graphic escape to CP310. -func decode(b []byte) string { +func (cp *charset) Decode(b []byte) string { runes := make([]rune, 0, len(b)) var escape bool for i := range b { if escape { escape = false - if cp310ToUnicode[b[i]] != '�' { - runes = append(runes, cp310ToUnicode[b[i]]) + if cp.ge2u[b[i]] != '�' { + runes = append(runes, cp.ge2u[b[i]]) } else { runes = append(runes, 0x1A) // Unicode substitute } } else { // Enter graphic escape mode if necessary. - if b[i] == 0x08 { + if b[i] == cp.ge { escape = true continue } // Otherwise perform the mapping. - runes = append(runes, cp1047ToUnicode[b[i]]) + runes = append(runes, cp.e2u[b[i]]) } } @@ -153,7 +210,7 @@ func decode(b []byte) string { } // encode will convert a UTF-8 Go string into a CP1047 byte array. -func encode(s string) []byte { +func (cp *charset) Encode(s string) []byte { out := make([]byte, 0, len(s)) for len(s) > 0 { @@ -163,14 +220,14 @@ func encode(s string) []byte { break } - if int(r) < len(unicodeToCP1047) { - out = append(out, unicodeToCP1047[r]) - } else if v, ok := unicodeToCP310[r]; ok { + if int(r) < len(cp.u2e) { + out = append(out, cp.u2e[r]) + } else if v, ok := cp.u2ge[r]; ok { // include graphic escape character to switch to CP310 - out = append(out, 0x08, v) + out = append(out, cp.ge, v) } else { // replacement/substitute character - out = append(out, 0x3f) + out = append(out, cp.esub) } s = s[size:] } diff --git a/response.go b/response.go index 111ea57..5209179 100644 --- a/response.go +++ b/response.go @@ -160,7 +160,7 @@ func readFields(c net.Conn, fm fieldmap, cols int) (map[string]string, error) { if eor { // Finish the current field if infield { - value := decode(fieldval.Bytes()) + value := currentCodepage.Decode(fieldval.Bytes()) debugf("Field %d: %s\n", fieldpos, value) handleField(fieldpos, value, fm, values) } @@ -172,7 +172,7 @@ func readFields(c net.Conn, fm fieldmap, cols int) (map[string]string, error) { if b == 0x11 { // Finish the previous field, if necessary if infield { - value := decode(fieldval.Bytes()) + value := currentCodepage.Decode(fieldval.Bytes()) debugf("Field %d: %s\n", fieldpos, value) handleField(fieldpos, value, fm, values) } diff --git a/screen.go b/screen.go index d7640e6..74f8bd1 100644 --- a/screen.go +++ b/screen.go @@ -286,7 +286,7 @@ func showScreenInternal(screen Screen, values map[string]string, } } if content != "" { - b.Write(encode(content)) + b.Write(currentCodepage.Encode(content)) } // If a writable field, add it to the field map. We add 1 to bufaddr