Skip to content

Commit

Permalink
#722 Add CP278 EBCDIC code page.
Browse files Browse the repository at this point in the history
  • Loading branch information
yruslan committed Oct 29, 2024
1 parent 4d2f413 commit ab0bd07
Show file tree
Hide file tree
Showing 4 changed files with 84 additions and 0 deletions.
Original file line number Diff line number Diff line change
Expand Up @@ -47,6 +47,7 @@ object CodePage extends Logging {
case "cp00300" => new CodePage300 // This is the same as cp300
case "cp273" => new CodePage273
case "cp277" => new CodePage277
case "cp278" => new CodePage278
case "cp300" => new CodePage300
case "cp500" => new CodePage500
case "cp838" => new CodePage838
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,54 @@
/*
* Copyright 2018 ABSA Group Limited
*
* Licensed under the Apache License, Version 2.0 (the "License");
* you may not use this file except in compliance with the License.
* You may obtain a copy of the License at
*
* http://www.apache.org/licenses/LICENSE-2.0
*
* Unless required by applicable law or agreed to in writing, software
* distributed under the License is distributed on an "AS IS" BASIS,
* WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
* See the License for the specific language governing permissions and
* limitations under the License.
*/

package za.co.absa.cobrix.cobol.parser.encoding.codepage

/**
* EBCDIC code page 278 is used to represent characters of Finland and Sweden.
*/
class CodePage278 extends SingleByteCodePage(CodePage278.ebcdicToAsciiMapping) {
override def codePageShortName: String = "cp278"
}

object CodePage278 {
val ebcdicToAsciiMapping: Array[Char] = {
import EbcdicNonPrintable._

/* This is the EBCDIC Code Page 278 to ASCII conversion table
from https://en.wikibooks.org/wiki/Character_Encodings/Code_Tables/EBCDIC/EBCDIC_278 */
val ebcdic2ascii: Array[Char] = {
// Non-printable characters map used: http://www.pacsys.com/asciitab.htm
Array[Char](
c00, c01, c02, c03, spc, c09, spc, del, spc, spc, spc, c0b, c0c, ccr, c0e, c0f, // 0 - 15
c10, c11, c12, c13, spc, nel, c08, spc, c18, c19, spc, spc, c1c, c1d, c1e, c1f, // 16 - 31
spc, spc, spc, spc, spc, clf, c17, c1b, spc, spc, spc, spc, spc, c05, c06, c07, // 32 - 47
spc, spc, c16, spc, spc, spc, spc, c04, spc, spc, spc, spc, c14, c15, spc, c1a, // 48 - 63
' ', rsp, 'â', '{', 'à', 'á', 'ã', '}', 'ç', 'ñ', '§', '.', '<', '(', '+', '!', // 64 - 79
'&', '`', 'ê', 'ë', 'è', 'í', 'î', 'ï', 'ì', 'ß', '¤', 'Å', '*', ')', ';', '^', // 80 - 95
'-', '/', 'Â', '#', 'À', 'Á', 'Ã', '$', 'Ç', 'Ñ', 'ö', ',', '%', '_', '>', '?', // 96 - 111
'ø', bsh, 'Ê', 'Ë', 'È', 'Í', 'Î', 'Ï', 'Ì', 'é', ':', 'Ä', 'Ö', qts, '=', qtd, // 112 - 127
'Ø', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', '«', '»', 'ð', 'ý', 'þ', '±', // 128 - 143
'°', 'j', 'k', 'l', 'm', 'n', 'o', 'p', 'q', 'r', 'ª', 'º', 'æ', '¸', 'Æ', ']', // 144 - 159
'µ', 'ü', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', '¡', '¿', 'Ð', 'Ý', 'Þ', '®', // 160 - 175
'¢', '£', '¥', '·', '©', '[', '¶', '¼', '½', '¾', '¬', '|', '¯', '¨', '´', '×', // 176 - 191
'ä', 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', shy, 'ô', '¦', 'ò', 'ó', 'õ', // 192 - 207
'å', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', '¹', 'û', '~', 'ù', 'ú', 'ÿ', // 208 - 223
'É', '÷', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z', '²', 'Ô', '@', 'Ò', 'Ó', 'Õ', // 224 - 239
'0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '³', 'Û', 'Ü', 'Ù', 'Ú', spc) // 240 - 255
}
ebcdic2ascii
}
}
Original file line number Diff line number Diff line change
Expand Up @@ -128,6 +128,30 @@ class StringDecodersSpec extends AnyWordSpec {
assert(actual == expected)
}

"decode a CP278 string special characters" in {
val expected = " {Ä!~Ü^[ö¤ß¢§@äåæ¦ü}ÖÆØ$#\\] "
val bytes = Array(0x40, 0x43, 0x7B, 0x4F, 0xDC, 0xFC, 0x5F, 0xB5, 0x6A, 0x5A, 0x59,
0xB0, 0x4A, 0xEC, 0xC0, 0xD0, 0x9C, 0xCC, 0xA1, 0x47, 0x7C, 0x9E, 0x80, 0x67, 0x63,
0x71, 0x9F, 0x40).map(_.toByte)

val actual = decodeEbcdicString(bytes, KeepAll, new CodePage278, improvedNullDetection = false)

assert(actual == expected)
}

"decode a CP278 string example" in {
val expected = "Ångbåten är över sjön med färggranna blommor."

val bytes = Array(0x5B, 0x95, 0x87, 0x82, 0xD0, 0xA3, 0x85, 0x95, 0x40, 0xC0, 0x99, 0x40, 0x6A,
0xA5, 0x85, 0x99, 0x40, 0xA2, 0x91, 0x6A, 0x95, 0x40, 0x94, 0x85, 0x84, 0x40, 0x86, 0xC0,
0x99, 0x87, 0x87, 0x99, 0x81, 0x95, 0x95, 0x81, 0x40, 0x82, 0x93, 0x96, 0x94, 0x94, 0x96,
0x99, 0x4B).map(_.toByte)

val actual = decodeEbcdicString(bytes, KeepAll, new CodePage278, improvedNullDetection = false)

assert(actual == expected)
}

"decode a CP500 string special characters" in {
val expected = "âäàáãåçñ[.<(+!&éêëèíîïìß]$*);^-/ÂÄÀÁÃÅÇѦ,%_>?øÉÊËÈÍÎÏÌ`:#@'=\"Øabcdefghi«»ðýþ±°jklmnopqrªºæ¸Æ¤µ~stuvwxyz¡¿ÐÝÞ®¢£¥·©§¶¼½¾¬|¯¨´×{ABCDEFGHI\u00ADôöòóõ}JKLMNOPQR¹ûüùúÿ\\÷STUVWXYZ²ÔÖÒÓÕ0123456789³ÛÜÙÚ"
val bytes = Array(
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -52,6 +52,11 @@ class CodePageSingleByteSpec extends AnyFunSuite {
assert(codePage.codePageShortName == "cp277")
}

test("Ensure codepage 'cp278' gives the associated CodePage") {
val codePage = CodePage.getCodePageByName("cp278")
assert(codePage.codePageShortName == "cp278")
}

test("Ensure codepage 'cp300' gives the associated CodePage") {
val codePage = CodePage.getCodePageByName("cp300")
assert(codePage.codePageShortName == "cp300")
Expand Down

0 comments on commit ab0bd07

Please sign in to comment.