Fix broken EBCDIC conversions. Fully switch to CP1047<->UTF-8.
This commit is contained in:
1 parent
86d2a23adf
commit
904699bf1e
4 files changed
+117
-59
No files matched your search
@@ -24,6 +24,17 @@ Here's [a video introducing the library][introVideo] as well.
|
||||
|
||||
[introVideo]: https://www.youtube.com/watch?v=h9XTjup5W5U
|
||||
|
||||
A note on code pages
|
||||
--------------------
|
||||
|
||||
After careful consideration, I have decided that the code page we will support for EBCDIC is IBM CP1047.
|
||||
|
||||
In suite3270 (e.g. c3270/x3270), the default code page is what it calls "brackets". This is CP37 with the [, ], Ý, and ¨ characters swapped around. This ends up placing all four of those characters in the correct place for 1047 (and thus they will work correctly with go3270 by default). However, the ^ and ¬ characters are swapped relative to CP1047. (Or, more succinctly, you could say the suite3270 "brackets" codepage is CP1047 with the ^ and ¬ characters swapped back to where they are in CP37). If you plan on using the ^ and ¬ characters, run c/x3270 in proper 1047 mode, `c3270 -codepage 1047` or make it your default by setting the `c3270.codePage` resource to `1047` in your `.c3270pro` file, for example.
|
||||
|
||||
In Vista TN3270, "United States" is the default code page. This is CP1047 and will map 100% correctly.
|
||||
|
||||
In IBM PCOMM, CP37 is the default. For correct mapping of [, ], Ý, ¨, ^, and ¬, you must switch the session parameters from "037 United States" to "1047 United States".
|
||||
|
||||
3270 information
|
||||
----------------
|
||||
|
||||
|
||||
@@ -4,61 +4,106 @@
|
||||
|
||||
package go3270
|
||||
|
||||
// Each index in this array is the ASCII value, and the value at the index is
|
||||
// the corresponding EBCDIC (codepage 37) value.
|
||||
var ebcdic = []byte{
|
||||
0, 1, 2, 3, 55, 45, 46, 47, 22, 5, 37, 11, 12, 13, 14, 15, 16, 17, 18, 19,
|
||||
60, 61, 50, 38, 24, 25, 63, 39, 28, 29, 30, 31, 64, 90, 127, 123, 91, 108,
|
||||
80, 125, 77, 93, 92, 78, 107, 96, 75, 97, 240, 241, 242, 243, 244, 245,
|
||||
246, 247, 248, 249, 122, 94, 76, 126, 110, 111, 124, 193, 194, 195, 196,
|
||||
197, 198, 199, 200, 201, 209, 210, 211, 212, 213, 214, 215, 216, 217, 226,
|
||||
227, 228, 229, 230, 231, 232, 233, 74, 224, 90, 95, 109, 121, 129, 130,
|
||||
131, 132, 133, 134, 135, 136, 137, 145, 146, 147, 148, 149, 150, 151, 152,
|
||||
153, 162, 163, 164, 165, 166, 167, 168, 169, 192, 106, 208, 161, 7, 32,
|
||||
33, 34, 35, 36, 21, 6, 23, 40, 41, 42, 43, 44, 9, 10, 27, 48, 49, 26, 51,
|
||||
52, 53, 54, 8, 56, 57, 58, 59, 4, 20, 62, 225, 65, 66, 67, 68, 69, 70, 71,
|
||||
72, 73, 81, 82, 83, 84, 85, 86, 87, 88, 89, 98, 99, 100, 101, 102, 103,
|
||||
104, 105, 112, 113, 114, 115, 116, 117, 118, 119, 120, 128, 138, 139, 140,
|
||||
141, 142, 143, 144, 154, 155, 156, 157, 158, 159, 160, 170, 171, 172, 173,
|
||||
174, 175, 176, 177, 178, 179, 180, 181, 182, 183, 184, 185, 186, 187, 188,
|
||||
189, 190, 191, 202, 203, 204, 205, 206, 207, 218, 219, 220, 221, 222, 223,
|
||||
234, 235, 236, 237, 238, 239, 250, 251, 252, 253, 254, 255}
|
||||
import (
|
||||
"unicode/utf8"
|
||||
)
|
||||
|
||||
// Each index in this array is the EBCDIC (codepage 37) value, and the value
|
||||
// at the index is the corresponding ASCII value.
|
||||
var ascii = []byte{
|
||||
0, 1, 2, 3, 156, 9, 134, 127, 151, 141, 142, 11, 12, 13, 14, 15,
|
||||
16, 17, 18, 19, 157, 133, 8, 135, 24, 25, 146, 143, 28, 29, 30, 31,
|
||||
128, 129, 130, 131, 132, 10, 23, 27, 136, 137, 138, 139, 140, 5, 6, 7,
|
||||
144, 145, 22, 147, 148, 149, 150, 4, 152, 153, 154, 155, 20, 21, 158, 26,
|
||||
32, 160, 161, 162, 163, 164, 165, 166, 167, 168, 91, 46, 60, 40, 43, 33,
|
||||
38, 169, 170, 171, 172, 173, 174, 175, 176, 177, 33, 36, 42, 41, 59, 94,
|
||||
45, 47, 178, 179, 180, 181, 182, 183, 184, 185, 124, 44, 37, 95, 62, 63,
|
||||
186, 187, 188, 189, 190, 191, 192, 193, 194, 96, 58, 35, 64, 39, 61, 34,
|
||||
195, 97, 98, 99, 100, 101, 102, 103, 104, 105, 196, 197, 198, 199, 200,
|
||||
201, 202, 106, 107, 108, 109, 110, 111, 112, 113, 114, 203, 204, 205, 206,
|
||||
207, 208, 209, 126, 115, 116, 117, 118, 119, 120, 121, 122, 210, 211, 212,
|
||||
213, 214, 215, 216, 217, 218, 219, 220, 221, 222, 223, 224, 225, 226, 227,
|
||||
228, 229, 230, 231, 123, 65, 66, 67, 68, 69, 70, 71, 72, 73, 232, 233,
|
||||
234, 235, 236, 237, 125, 74, 75, 76, 77, 78, 79, 80, 81, 82, 238, 239,
|
||||
240, 241, 242, 243, 92, 159, 83, 84, 85, 86, 87, 88, 89, 90, 244, 245,
|
||||
246, 247, 248, 249, 48, 49, 50, 51, 52, 53, 54, 55, 56, 57, 250, 251, 252,
|
||||
253, 254, 255}
|
||||
// After careful consideration, I have decided that the code page we will
|
||||
// support for EBCDIC is IBM CP 1047.
|
||||
//
|
||||
// In suite3270 (e.g. c3270/x3270), the default code page is what it calls
|
||||
// "brackets". This is CP37 with the [, ], Ý, and ¨ characters swapped around.
|
||||
// This ends up placing all four of those characters in the correct place for
|
||||
// 1047 (and thus they will all work correctly with go3270 by default).
|
||||
// HOWEVER, the ^ and ¬ characters are swapped relative to CP1047. (Or, more
|
||||
// succinctly, you could say the suite3270 "brackets" codepage is CP1047 with
|
||||
// the ^ and ¬ characters swapped back to where they are in CP37). If you plan
|
||||
// on using the ^ and ¬ characters, run c/x3270 in proper 1047 mode,
|
||||
// `c3270-codepage 1047` or make it your default by setting the
|
||||
// `c3270.codePage` resource to `1047` in your `.c3270pro` file, for example.
|
||||
//
|
||||
// In Vista TN3270, "United States" is the default code page. This is CP1047
|
||||
// and will map 100% correctly.
|
||||
//
|
||||
// In IBM PCOMM, CP37 is the default. For correct mapping of [, ], Ý, ¨, ^,
|
||||
// and ¬, you must switch the session parameters from "037 United States" to
|
||||
// "1047 United States".
|
||||
|
||||
// a2e converts the input byte array, a, from ASCII to EBCDIC.
|
||||
func a2e(a []byte) []byte {
|
||||
result := make([]byte, len(a))
|
||||
for i := 0; i < len(a); i++ {
|
||||
result[i] = ebcdic[a[i]]
|
||||
}
|
||||
return result
|
||||
// IBM CP 1047 <-> Unicode mappings from:
|
||||
// https://raw.githubusercontent.com/unicode-org/icu-data/refs/heads/main/charset/data/ucm/glibc-IBM1047-2.1.2.ucm
|
||||
|
||||
var cp1047ToUnicode []rune = []rune{
|
||||
/* x0 x1 x2 x3 x4 x5 x6 x7 x8 x9 xA xB xC xD xE xF */
|
||||
/* 0x */ 0x00, 0x01, 0x02, 0x03, 0x9C, 0x09, 0x86, 0x7F, 0x97, 0x8D, 0x8E, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F,
|
||||
/* 1x */ 0x10, 0x11, 0x12, 0x13, 0x9D, 0x85, 0x08, 0x87, 0x18, 0x19, 0x92, 0x8F, 0x1C, 0x1D, 0x1E, 0x1F,
|
||||
/* 2x */ 0x80, 0x81, 0x82, 0x83, 0x84, 0x0A, 0x17, 0x1B, 0x88, 0x89, 0x8A, 0x8B, 0x8C, 0x05, 0x06, 0x07,
|
||||
/* 3x */ 0x90, 0x91, 0x16, 0x93, 0x94, 0x95, 0x96, 0x04, 0x98, 0x99, 0x9A, 0x9B, 0x14, 0x15, 0x9E, 0x1A,
|
||||
/* 4x */ 0x20, 0xA0, 0xE2, 0xE4, 0xE0, 0xE1, 0xE3, 0xE5, 0xE7, 0xF1, 0xA2, 0x2E, 0x3C, 0x28, 0x2B, 0x7C,
|
||||
/* 5x */ 0x26, 0xE9, 0xEA, 0xEB, 0xE8, 0xED, 0xEE, 0xEF, 0xEC, 0xDF, 0x21, 0x24, 0x2A, 0x29, 0x3B, 0x5E,
|
||||
/* 6x */ 0x2D, 0x2F, 0xC2, 0xC4, 0xC0, 0xC1, 0xC3, 0xC5, 0xC7, 0xD1, 0xA6, 0x2C, 0x25, 0x5F, 0x3E, 0x3F,
|
||||
/* 7x */ 0xF8, 0xC9, 0xCA, 0xCB, 0xC8, 0xCD, 0xCE, 0xCF, 0xCC, 0x60, 0x3A, 0x23, 0x40, 0x27, 0x3D, 0x22,
|
||||
/* 8x */ 0xD8, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, 0x67, 0x68, 0x69, 0xAB, 0xBB, 0xF0, 0xFD, 0xFE, 0xB1,
|
||||
/* 9x */ 0xB0, 0x6A, 0x6B, 0x6C, 0x6D, 0x6E, 0x6F, 0x70, 0x71, 0x72, 0xAA, 0xBA, 0xE6, 0xB8, 0xC6, 0xA4,
|
||||
/* Ax */ 0xB5, 0x7E, 0x73, 0x74, 0x75, 0x76, 0x77, 0x78, 0x79, 0x7A, 0xA1, 0xBF, 0xD0, 0x5B, 0xDE, 0xAE,
|
||||
/* Bx */ 0xAC, 0xA3, 0xA5, 0xB7, 0xA9, 0xA7, 0xB6, 0xBC, 0xBD, 0xBE, 0xDD, 0xA8, 0xAF, 0x5D, 0xB4, 0xD7,
|
||||
/* Cx */ 0x7B, 0x41, 0x42, 0x43, 0x44, 0x45, 0x46, 0x47, 0x48, 0x49, 0xAD, 0xF4, 0xF6, 0xF2, 0xF3, 0xF5,
|
||||
/* Dx */ 0x7D, 0x4A, 0x4B, 0x4C, 0x4D, 0x4E, 0x4F, 0x50, 0x51, 0x52, 0xB9, 0xFB, 0xFC, 0xF9, 0xFA, 0xFF,
|
||||
/* Ex */ 0x5C, 0xF7, 0x53, 0x54, 0x55, 0x56, 0x57, 0x58, 0x59, 0x5A, 0xB2, 0xD4, 0xD6, 0xD2, 0xD3, 0xD5,
|
||||
/* Fx */ 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, 0x38, 0x39, 0xB3, 0xDB, 0xDC, 0xD9, 0xDA, 0x9F,
|
||||
}
|
||||
|
||||
// e2a converts the input byte array, e, from EBCDIC to ASCII
|
||||
func e2a(e []byte) []byte {
|
||||
result := make([]byte, len(e))
|
||||
for i := 0; i < len(e); i++ {
|
||||
result[i] = ascii[e[i]]
|
||||
var unicodeToCP1047 []byte = []byte{
|
||||
/* x0 x1 x2 x3 x4 x5 x6 x7 x8 x9 xA xB xC xD xE xF */
|
||||
/* 0x */ 0x00, 0x01, 0x02, 0x03, 0x37, 0x2D, 0x2E, 0x2F, 0x16, 0x05, 0x25, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F,
|
||||
/* 1x */ 0x10, 0x11, 0x12, 0x13, 0x3C, 0x3D, 0x32, 0x26, 0x18, 0x19, 0x3F, 0x27, 0x1C, 0x1D, 0x1E, 0x1F,
|
||||
/* 2x */ 0x40, 0x5A, 0x7F, 0x7B, 0x5B, 0x6C, 0x50, 0x7D, 0x4D, 0x5D, 0x5C, 0x4E, 0x6B, 0x60, 0x4B, 0x61,
|
||||
/* 3x */ 0xF0, 0xF1, 0xF2, 0xF3, 0xF4, 0xF5, 0xF6, 0xF7, 0xF8, 0xF9, 0x7A, 0x5E, 0x4C, 0x7E, 0x6E, 0x6F,
|
||||
/* 4x */ 0x7C, 0xC1, 0xC2, 0xC3, 0xC4, 0xC5, 0xC6, 0xC7, 0xC8, 0xC9, 0xD1, 0xD2, 0xD3, 0xD4, 0xD5, 0xD6,
|
||||
/* 5x */ 0xD7, 0xD8, 0xD9, 0xE2, 0xE3, 0xE4, 0xE5, 0xE6, 0xE7, 0xE8, 0xE9, 0xAD, 0xE0, 0xBD, 0x5F, 0x6D,
|
||||
/* 6x */ 0x79, 0x81, 0x82, 0x83, 0x84, 0x85, 0x86, 0x87, 0x88, 0x89, 0x91, 0x92, 0x93, 0x94, 0x95, 0x96,
|
||||
/* 7x */ 0x97, 0x98, 0x99, 0xA2, 0xA3, 0xA4, 0xA5, 0xA6, 0xA7, 0xA8, 0xA9, 0xC0, 0x4F, 0xD0, 0xA1, 0x07,
|
||||
/* 8x */ 0x20, 0x21, 0x22, 0x23, 0x24, 0x15, 0x06, 0x17, 0x28, 0x29, 0x2A, 0x2B, 0x2C, 0x09, 0x0A, 0x1B,
|
||||
/* 9x */ 0x30, 0x31, 0x1A, 0x33, 0x34, 0x35, 0x36, 0x08, 0x38, 0x39, 0x3A, 0x3B, 0x04, 0x14, 0x3E, 0xFF,
|
||||
/* Ax */ 0x41, 0xAA, 0x4A, 0xB1, 0x9F, 0xB2, 0x6A, 0xB5, 0xBB, 0xB4, 0x9A, 0x8A, 0xB0, 0xCA, 0xAF, 0xBC,
|
||||
/* Bx */ 0x90, 0x8F, 0xEA, 0xFA, 0xBE, 0xA0, 0xB6, 0xB3, 0x9D, 0xDA, 0x9B, 0x8B, 0xB7, 0xB8, 0xB9, 0xAB,
|
||||
/* Cx */ 0x64, 0x65, 0x62, 0x66, 0x63, 0x67, 0x9E, 0x68, 0x74, 0x71, 0x72, 0x73, 0x78, 0x75, 0x76, 0x77,
|
||||
/* Dx */ 0xAC, 0x69, 0xED, 0xEE, 0xEB, 0xEF, 0xEC, 0xBF, 0x80, 0xFD, 0xFE, 0xFB, 0xFC, 0xBA, 0xAE, 0x59,
|
||||
/* Ex */ 0x44, 0x45, 0x42, 0x46, 0x43, 0x47, 0x9C, 0x48, 0x54, 0x51, 0x52, 0x53, 0x58, 0x55, 0x56, 0x57,
|
||||
/* Fx */ 0x8C, 0x49, 0xCD, 0xCE, 0xCB, 0xCF, 0xCC, 0xE1, 0x70, 0xDD, 0xDE, 0xDB, 0xDC, 0x8D, 0x8E, 0xDF,
|
||||
}
|
||||
|
||||
// decode will convert a CP1047 byte array into a UTF-8 Go string.
|
||||
func decode(b []byte) string {
|
||||
// There is complete 1:1 mapping of all 8-bit values
|
||||
runes := make([]rune, len(b))
|
||||
for i := range b {
|
||||
runes[i] = cp1047ToUnicode[b[i]]
|
||||
}
|
||||
return result
|
||||
return string(runes) // conversion to UTF-8 is automatic
|
||||
}
|
||||
|
||||
// encode will convert a UTF-8 Go string into a CP1047 byte array.
|
||||
func encode(s string) []byte {
|
||||
// Output CP1047 bytes will be no more than input UTF-8 bytes
|
||||
out := make([]byte, len(s))
|
||||
n := 0
|
||||
|
||||
for len(s) > 0 {
|
||||
r, size := utf8.DecodeRuneInString(s)
|
||||
if r == utf8.RuneError {
|
||||
debugf("invalid UTF-8 encoding detected, aborting conversion")
|
||||
break
|
||||
}
|
||||
|
||||
if int(r) < len(unicodeToCP1047) {
|
||||
out[n] = unicodeToCP1047[r]
|
||||
} else {
|
||||
// replacement/substitute character
|
||||
out[n] = 0x3f
|
||||
}
|
||||
n++
|
||||
s = s[size:]
|
||||
}
|
||||
|
||||
return out[:n]
|
||||
}
|
||||
+8
-6
@@ -160,8 +160,9 @@ func readFields(c net.Conn, fm fieldmap, cols int) (map[string]string, error) {
|
||||
if eor {
|
||||
// Finish the current field
|
||||
if infield {
|
||||
debugf("Field %d: %s\n", fieldpos, e2a(fieldval.Bytes()))
|
||||
handleField(fieldpos, fieldval.Bytes(), fm, values)
|
||||
value := decode(fieldval.Bytes())
|
||||
debugf("Field %d: %s\n", fieldpos, value)
|
||||
handleField(fieldpos, value, fm, values)
|
||||
}
|
||||
|
||||
return values, nil
|
||||
@@ -171,8 +172,9 @@ func readFields(c net.Conn, fm fieldmap, cols int) (map[string]string, error) {
|
||||
if b == 0x11 {
|
||||
// Finish the previous field, if necessary
|
||||
if infield {
|
||||
debugf("Field %d: %s\n", fieldpos, e2a(fieldval.Bytes()))
|
||||
handleField(fieldpos, fieldval.Bytes(), fm, values)
|
||||
value := decode(fieldval.Bytes())
|
||||
debugf("Field %d: %s\n", fieldpos, value)
|
||||
handleField(fieldpos, value, fm, values)
|
||||
}
|
||||
// Start a new field
|
||||
infield = true
|
||||
@@ -194,7 +196,7 @@ func readFields(c net.Conn, fm fieldmap, cols int) (map[string]string, error) {
|
||||
}
|
||||
}
|
||||
|
||||
func handleField(addr int, value []byte, fm fieldmap, values map[string]string) bool {
|
||||
func handleField(addr int, value string, fm fieldmap, values map[string]string) bool {
|
||||
name, ok := fm[addr]
|
||||
|
||||
// Field is not present in the fieldmap
|
||||
@@ -203,7 +205,7 @@ func handleField(addr int, value []byte, fm fieldmap, values map[string]string)
|
||||
}
|
||||
|
||||
// Otherwise, populate the value
|
||||
values[name] = string(e2a(value))
|
||||
values[name] = value
|
||||
return true
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user