237 lines
12 KiB
Go
237 lines
12 KiB
Go
// This file is part of https://github.com/racingmars/go3270/
|
||
// Copyright 2020 by Matthew R. Wilson, licensed under the MIT license. See
|
||
// LICENSE in the project root for license information.
|
||
|
||
package go3270
|
||
|
||
import (
|
||
"unicode/utf8"
|
||
)
|
||
|
||
// Implementations of Charset provide EBCDIC<->UTF-8 translation.
|
||
type Charset interface {
|
||
// Decode converts a slice of EBCDIC bytes into a UTF-8 string.
|
||
Decode(e []byte) string
|
||
|
||
// Encode converts a UTF-8 string into a slice of EBCDIC bytes.
|
||
Encode(s string) []byte
|
||
}
|
||
|
||
// After careful consideration, I have decided that the default code page we
|
||
// will support for EBCDIC is IBM CP 1047. Other code pages may be globally
|
||
// selected with the SetCodepage() function.
|
||
//
|
||
// In suite3270 (e.g. c3270/x3270), the default code page is what it calls
|
||
// "brackets". This is CP37 with the [, ], Ý, and ¨ characters swapped around.
|
||
// This ends up placing all four of those characters in the correct place for
|
||
// 1047 (and thus they will all work correctly with go3270 by default).
|
||
// HOWEVER, the ^ and ¬ characters are swapped relative to CP1047. (Or, more
|
||
// succinctly, you could say the suite3270 "brackets" codepage is CP1047 with
|
||
// the ^ and ¬ characters swapped back to where they are in CP37). If you plan
|
||
// on using the ^ and ¬ characters, run c/x3270 in proper 1047 mode,
|
||
// `c3270-codepage 1047` or make it your default by setting the
|
||
// `c3270.codePage` resource to `1047` in your `.c3270pro` file, for example.
|
||
//
|
||
// In Vista TN3270, "United States" is the default code page. This is CP1047
|
||
// and will map 100% correctly.
|
||
//
|
||
// In IBM PCOMM, CP37 is the default. For correct mapping of [, ], Ý, ¨, ^,
|
||
// and ¬, you must switch the session parameters from "037 United States" to
|
||
// "1047 United States".
|
||
var currentCodepage Charset = Codepage1047()
|
||
|
||
// SetCodepage sets the codepage/character set that go3270 uses. This is a
|
||
// global setting, so if you're expecting clients to be configured to use a
|
||
// character set other than go3270's default, cp1047, you should probably set
|
||
// this during your application initialization and then leave it unchanged
|
||
// after. This is _not_ a per-connection setting.
|
||
func SetCodepage(cs Charset) {
|
||
currentCodepage = cs
|
||
}
|
||
|
||
// Internal implementation of the Charset interface we'll use for the codepage
|
||
// support we provide.
|
||
type charset struct {
|
||
// EBCDIC byte to Unicode code point for bytes 0x00-0xFF
|
||
e2u []rune
|
||
|
||
// Unicode code point to EBCDIC byte for codepoints 0x00-0xFF
|
||
u2e []byte
|
||
|
||
// Map of Unicode code points to EBCDIC bytes for codepoints >0xFF
|
||
highu2e map[rune]byte
|
||
|
||
// The EBCDIC substitute character to use if there is no EBCDIC character
|
||
// for the requested Unicode code point (typically 0x3F).
|
||
esub byte
|
||
|
||
// The "graphic escape" EBCDIC byte (is it ever anything other than 0x0E?)
|
||
ge byte
|
||
|
||
// Graphic escape codepage EBCDIC byte to Unicode code point for bytes
|
||
// 0x00-0xFF. Use rune '�' for unmapped bytes.
|
||
ge2u []rune
|
||
|
||
// Map of Unicode code points to graphic escape EBCDIC bytes.
|
||
u2ge map[rune]byte
|
||
}
|
||
|
||
// IBM CP 1047 <-> Unicode mappings from:
|
||
// https://raw.githubusercontent.com/unicode-org/icu-data/refs/heads/main/charset/data/ucm/glibc-IBM1047-2.1.2.ucm
|
||
func Codepage1047() Charset {
|
||
return &charset{
|
||
e2u: []rune{
|
||
/* x0 x1 x2 x3 x4 x5 x6 x7 x8 x9 xA xB xC xD xE xF */
|
||
/* 0x */ 0x00, 0x01, 0x02, 0x03, 0x9C, 0x09, 0x86, 0x7F, 0x97, 0x8D, 0x8E, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F,
|
||
/* 1x */ 0x10, 0x11, 0x12, 0x13, 0x9D, 0x85, 0x08, 0x87, 0x18, 0x19, 0x92, 0x8F, 0x1C, 0x1D, 0x1E, 0x1F,
|
||
/* 2x */ 0x80, 0x81, 0x82, 0x83, 0x84, 0x0A, 0x17, 0x1B, 0x88, 0x89, 0x8A, 0x8B, 0x8C, 0x05, 0x06, 0x07,
|
||
/* 3x */ 0x90, 0x91, 0x16, 0x93, 0x94, 0x95, 0x96, 0x04, 0x98, 0x99, 0x9A, 0x9B, 0x14, 0x15, 0x9E, 0x1A,
|
||
/* 4x */ 0x20, 0xA0, 0xE2, 0xE4, 0xE0, 0xE1, 0xE3, 0xE5, 0xE7, 0xF1, 0xA2, 0x2E, 0x3C, 0x28, 0x2B, 0x7C,
|
||
/* 5x */ 0x26, 0xE9, 0xEA, 0xEB, 0xE8, 0xED, 0xEE, 0xEF, 0xEC, 0xDF, 0x21, 0x24, 0x2A, 0x29, 0x3B, 0x5E,
|
||
/* 6x */ 0x2D, 0x2F, 0xC2, 0xC4, 0xC0, 0xC1, 0xC3, 0xC5, 0xC7, 0xD1, 0xA6, 0x2C, 0x25, 0x5F, 0x3E, 0x3F,
|
||
/* 7x */ 0xF8, 0xC9, 0xCA, 0xCB, 0xC8, 0xCD, 0xCE, 0xCF, 0xCC, 0x60, 0x3A, 0x23, 0x40, 0x27, 0x3D, 0x22,
|
||
/* 8x */ 0xD8, 0x61, 0x62, 0x63, 0x64, 0x65, 0x66, 0x67, 0x68, 0x69, 0xAB, 0xBB, 0xF0, 0xFD, 0xFE, 0xB1,
|
||
/* 9x */ 0xB0, 0x6A, 0x6B, 0x6C, 0x6D, 0x6E, 0x6F, 0x70, 0x71, 0x72, 0xAA, 0xBA, 0xE6, 0xB8, 0xC6, 0xA4,
|
||
/* Ax */ 0xB5, 0x7E, 0x73, 0x74, 0x75, 0x76, 0x77, 0x78, 0x79, 0x7A, 0xA1, 0xBF, 0xD0, 0x5B, 0xDE, 0xAE,
|
||
/* Bx */ 0xAC, 0xA3, 0xA5, 0xB7, 0xA9, 0xA7, 0xB6, 0xBC, 0xBD, 0xBE, 0xDD, 0xA8, 0xAF, 0x5D, 0xB4, 0xD7,
|
||
/* Cx */ 0x7B, 0x41, 0x42, 0x43, 0x44, 0x45, 0x46, 0x47, 0x48, 0x49, 0xAD, 0xF4, 0xF6, 0xF2, 0xF3, 0xF5,
|
||
/* Dx */ 0x7D, 0x4A, 0x4B, 0x4C, 0x4D, 0x4E, 0x4F, 0x50, 0x51, 0x52, 0xB9, 0xFB, 0xFC, 0xF9, 0xFA, 0xFF,
|
||
/* Ex */ 0x5C, 0xF7, 0x53, 0x54, 0x55, 0x56, 0x57, 0x58, 0x59, 0x5A, 0xB2, 0xD4, 0xD6, 0xD2, 0xD3, 0xD5,
|
||
/* Fx */ 0x30, 0x31, 0x32, 0x33, 0x34, 0x35, 0x36, 0x37, 0x38, 0x39, 0xB3, 0xDB, 0xDC, 0xD9, 0xDA, 0x9F,
|
||
},
|
||
u2e: []byte{
|
||
/* x0 x1 x2 x3 x4 x5 x6 x7 x8 x9 xA xB xC xD xE xF */
|
||
/* 0x */ 0x00, 0x01, 0x02, 0x03, 0x37, 0x2D, 0x2E, 0x2F, 0x16, 0x05, 0x25, 0x0B, 0x0C, 0x0D, 0x0E, 0x0F,
|
||
/* 1x */ 0x10, 0x11, 0x12, 0x13, 0x3C, 0x3D, 0x32, 0x26, 0x18, 0x19, 0x3F, 0x27, 0x1C, 0x1D, 0x1E, 0x1F,
|
||
/* 2x */ 0x40, 0x5A, 0x7F, 0x7B, 0x5B, 0x6C, 0x50, 0x7D, 0x4D, 0x5D, 0x5C, 0x4E, 0x6B, 0x60, 0x4B, 0x61,
|
||
/* 3x */ 0xF0, 0xF1, 0xF2, 0xF3, 0xF4, 0xF5, 0xF6, 0xF7, 0xF8, 0xF9, 0x7A, 0x5E, 0x4C, 0x7E, 0x6E, 0x6F,
|
||
/* 4x */ 0x7C, 0xC1, 0xC2, 0xC3, 0xC4, 0xC5, 0xC6, 0xC7, 0xC8, 0xC9, 0xD1, 0xD2, 0xD3, 0xD4, 0xD5, 0xD6,
|
||
/* 5x */ 0xD7, 0xD8, 0xD9, 0xE2, 0xE3, 0xE4, 0xE5, 0xE6, 0xE7, 0xE8, 0xE9, 0xAD, 0xE0, 0xBD, 0x5F, 0x6D,
|
||
/* 6x */ 0x79, 0x81, 0x82, 0x83, 0x84, 0x85, 0x86, 0x87, 0x88, 0x89, 0x91, 0x92, 0x93, 0x94, 0x95, 0x96,
|
||
/* 7x */ 0x97, 0x98, 0x99, 0xA2, 0xA3, 0xA4, 0xA5, 0xA6, 0xA7, 0xA8, 0xA9, 0xC0, 0x4F, 0xD0, 0xA1, 0x07,
|
||
/* 8x */ 0x20, 0x21, 0x22, 0x23, 0x24, 0x15, 0x06, 0x17, 0x28, 0x29, 0x2A, 0x2B, 0x2C, 0x09, 0x0A, 0x1B,
|
||
/* 9x */ 0x30, 0x31, 0x1A, 0x33, 0x34, 0x35, 0x36, 0x08, 0x38, 0x39, 0x3A, 0x3B, 0x04, 0x14, 0x3E, 0xFF,
|
||
/* Ax */ 0x41, 0xAA, 0x4A, 0xB1, 0x9F, 0xB2, 0x6A, 0xB5, 0xBB, 0xB4, 0x9A, 0x8A, 0xB0, 0xCA, 0xAF, 0xBC,
|
||
/* Bx */ 0x90, 0x8F, 0xEA, 0xFA, 0xBE, 0xA0, 0xB6, 0xB3, 0x9D, 0xDA, 0x9B, 0x8B, 0xB7, 0xB8, 0xB9, 0xAB,
|
||
/* Cx */ 0x64, 0x65, 0x62, 0x66, 0x63, 0x67, 0x9E, 0x68, 0x74, 0x71, 0x72, 0x73, 0x78, 0x75, 0x76, 0x77,
|
||
/* Dx */ 0xAC, 0x69, 0xED, 0xEE, 0xEB, 0xEF, 0xEC, 0xBF, 0x80, 0xFD, 0xFE, 0xFB, 0xFC, 0xBA, 0xAE, 0x59,
|
||
/* Ex */ 0x44, 0x45, 0x42, 0x46, 0x43, 0x47, 0x9C, 0x48, 0x54, 0x51, 0x52, 0x53, 0x58, 0x55, 0x56, 0x57,
|
||
/* Fx */ 0x8C, 0x49, 0xCD, 0xCE, 0xCB, 0xCF, 0xCC, 0xE1, 0x70, 0xDD, 0xDE, 0xDB, 0xDC, 0x8D, 0x8E, 0xDF,
|
||
},
|
||
highu2e: map[rune]byte{},
|
||
esub: 0x3f,
|
||
ge: 0x08,
|
||
ge2u: cp310ToUnicode,
|
||
u2ge: unicodeToCP310,
|
||
}
|
||
}
|
||
|
||
// Certain characters are supported in the "graphic escape" CP310. These are
|
||
// arbitrary Unicode code points, so we will look them up via a map. For
|
||
// simplicity of our mapping implementation, we will not support the italic
|
||
// underlined A-Z characters that require combining characters.
|
||
//
|
||
// We will share this map among all of the codepages that we provide
|
||
// implementations for.
|
||
//
|
||
// https://public.dhe.ibm.com/software/globalization/gcoc/attachments/CP00310.pdf
|
||
var unicodeToCP310 = map[rune]byte{
|
||
'◊': 0x70, '⋄': 0x70, '◆': 0x70, '∧': 0x71, '⋀': 0x71, '¨': 0x72,
|
||
'⌻': 0x73, '⍸': 0x74, '⍷': 0x75, '⊢': 0x76, '⊣': 0x77, '∨': 0x78,
|
||
'∼': 0x80, '║': 0x81, '═': 0x82, '⎸': 0x83, '⎹': 0x84, '│': 0x85,
|
||
'⎥': 0x85, '↑': 0x8A, '↓': 0x8B, '≤': 0x8C, '⌈': 0x8D, '⌊': 0x8E,
|
||
'→': 0x8F, '⎕': 0x90, '▌': 0x91, '▐': 0x92, '▀': 0x93, '▄': 0x94,
|
||
'█': 0x95, '⊃': 0x9A, '⊂': 0x9B, '⌑': 0x9C, '¤': 0x9C, '○': 0x9D,
|
||
'±': 0x9E, '←': 0x9F, '¯': 0xA0, '‾': 0xA0, '°': 0xA1, '─': 0xA2,
|
||
'∙': 0xA3, '•': 0xA3, 'ₙ': 0xA4, '∩': 0xAA, '⋂': 0xAA, '∪': 0xAB,
|
||
'⋃': 0xAB, '⊥': 0xAC, '≥': 0xAE, '∘': 0xAF, '⍺': 0xB0, 'α': 0xB0,
|
||
'∊': 0xB1, '∈': 0xB1, 'ε': 0xB1, '⍳': 0xB2, 'ι': 0xB2, '⍴': 0xB3,
|
||
'ρ': 0xB3, '⍵': 0xB4, 'ω': 0xB4, '×': 0xB6, '∖': 0xB7, '÷': 0xB8,
|
||
'∇': 0xBA, '∆': 0xBB, '⊤': 0xBC, '≠': 0xBE, '∣': 0xBF, '⁽': 0xC1,
|
||
'⁺': 0xC2, '■': 0xC3, '∎': 0xC3, '└': 0xC4, '┌': 0xC5, '├': 0xC6,
|
||
'┴': 0xC7, '⍲': 0xCA, '⍱': 0xCB, '⌷': 0xCC, '⌽': 0xCD, '⍂': 0xCE,
|
||
'⍉': 0xCF, '⁾': 0xD1, '⁻': 0xD2, '┼': 0xD3, '┘': 0xD4, '┐': 0xD5,
|
||
'┤': 0xD6, '┬': 0xD7, '¶': 0xD8, '⌶': 0xDA, 'ǃ': 0xDB, '⍒': 0xDC,
|
||
'⍋': 0xDD, '⍞': 0xDE, '⍝': 0xDF, '≡': 0xE0, '₁': 0xE1, '₂': 0xE2,
|
||
'₃': 0xE3, '⍤': 0xE4, '⍥': 0xE5, '⍪': 0xE6, '€': 0xE7, '⌿': 0xEA,
|
||
'⍀': 0xEB, '∵': 0xEC, '⊖': 0xED, '⌹': 0xEE, '⍕': 0xEF, '⁰': 0xF0,
|
||
'¹': 0xF1, '²': 0xF2, '³': 0xF3, '⁴': 0xF4, '⁵': 0xF5, '⁶': 0xF6,
|
||
'⁷': 0xF7, '⁸': 0xF8, '⁹': 0xF9, '⍫': 0xFB, '⍙': 0xFC, '⍟': 0xFD,
|
||
'⍎': 0xFE,
|
||
}
|
||
|
||
// '�', the Unicode replacement character, is used as a placeholder in byte
|
||
// positions that are not assigned in this codepage.
|
||
var cp310ToUnicode = []rune{
|
||
/* x0 x1 x2 x3 x4 x5 x6 x7 x8 x9 xA xB xC xD xE xF */
|
||
/* 0x */ '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�',
|
||
/* 1x */ '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�',
|
||
/* 2x */ '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�',
|
||
/* 3x */ '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�',
|
||
/* 4x */ '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�',
|
||
/* 5x */ '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�',
|
||
/* 6x */ '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�', '�',
|
||
/* 7x */ '◊', '∧', '¨', '⌻', '⍸', '⍷', '⊢', '⊣', '∨', '�', '�', '�', '�', '�', '�', '�',
|
||
/* 8x */ '∼', '║', '═', '⎸', '⎹', '⎥', '�', '�', '�', '�', '↑', '↓', '≤', '⌈', '⌊', '→',
|
||
/* 9x */ '⎕', '▌', '▐', '▀', '▄', '█', '�', '�', '�', '�', '⊃', '⊂', '⌑', '○', '±', '←',
|
||
/* Ax */ '‾', '°', '─', '•', 'ₙ', '�', '�', '�', '�', '�', '∩', '⋃', '⊥', '�', '≥', '∘',
|
||
/* Bx */ '⍺', '∈', '⍳', '⍴', 'ω', '�', '×', '∖', '÷', '�', '∇', '∆', '⊤', '�', '≠', '∣',
|
||
/* Cx */ '�', '⁽', '⁺', '■', '└', '┌', '├', '┴', '�', '�', '⍲', '⍱', '⌷', '⌽', '⍂', '⍉',
|
||
/* Dx */ '�', '⁾', '⁻', '┼', '┘', '┐', '┤', '┬', '¶', '�', '⌶', 'ǃ', '⍒', '⍋', '⍞', '⍝',
|
||
/* Ex */ '≡', '₁', '₂', '₃', '⍤', '⍥', '⍪', '€', '�', '�', '⌿', '⍀', '∵', '⊖', '⌹', '⍕',
|
||
/* Fx */ '⁰', '¹', '²', '³', '⁴', '⁵', '⁶', '⁷', '⁸', '⁹', '�', '⍫', '⍙', '⍟', '⍎', '�',
|
||
}
|
||
|
||
// decode will convert a CP1047 byte array into a UTF-8 Go string, handling
|
||
// graphic escape to CP310.
|
||
func (cp *charset) Decode(b []byte) string {
|
||
runes := make([]rune, 0, len(b))
|
||
var escape bool
|
||
for i := range b {
|
||
if escape {
|
||
escape = false
|
||
if cp.ge2u[b[i]] != '�' {
|
||
runes = append(runes, cp.ge2u[b[i]])
|
||
} else {
|
||
runes = append(runes, 0x1A) // Unicode substitute
|
||
}
|
||
} else {
|
||
// Enter graphic escape mode if necessary.
|
||
if b[i] == cp.ge {
|
||
escape = true
|
||
continue
|
||
}
|
||
// Otherwise perform the mapping.
|
||
runes = append(runes, cp.e2u[b[i]])
|
||
}
|
||
|
||
}
|
||
return string(runes) // conversion to UTF-8 is automatic
|
||
}
|
||
|
||
// encode will convert a UTF-8 Go string into a CP1047 byte array.
|
||
func (cp *charset) Encode(s string) []byte {
|
||
out := make([]byte, 0, len(s))
|
||
|
||
for len(s) > 0 {
|
||
r, size := utf8.DecodeRuneInString(s)
|
||
if r == utf8.RuneError {
|
||
debugf("invalid UTF-8 encoding detected, aborting conversion")
|
||
break
|
||
}
|
||
|
||
if int(r) < len(cp.u2e) {
|
||
out = append(out, cp.u2e[r])
|
||
} else if v, ok := cp.u2ge[r]; ok {
|
||
// include graphic escape character to switch to CP310
|
||
out = append(out, cp.ge, v)
|
||
} else {
|
||
// replacement/substitute character
|
||
out = append(out, cp.esub)
|
||
}
|
||
s = s[size:]
|
||
}
|
||
|
||
return out
|
||
}
|