실행 방법입니다. 외부 jar 는 필요 없습니다.
cd java-src\extension\14_encoding
javac -encoding UTF-8 *.java && java Main콘솔 한글이 깨지면 java -Dstdout.encoding=UTF-8 Main을 쓰거나, cmd 에서 먼저 chcp 65001을 실행합니다.
// Main.demo1_byteTable
String s = "한글A";
printRow("UTF-8", s, StandardCharsets.UTF_8);
printRow("EUC-KR", s, EUC_KR);
printRow("CP949", s, CP949);UTF-8 7바이트 ed959ceab88041 ← 한글은 3바이트씩, A 는 1바이트
EUC-KR 5바이트 c7d1b1db41 ← 한글은 2바이트씩
CP949 5바이트 c7d1b1db41 ← 완성형 한글은 EUC-KR 과 바이트가 같다// Main.demo2_brokenDecoding
byte[] utf8Bytes = "한글".getBytes(StandardCharsets.UTF_8);
String brokenByEucKr = new String(utf8Bytes, EUC_KR);
// ISO-8859-1 은 바이트를 1:1로 문자에 매핑해 정보 손실이 없다.
String passthrough = new String(utf8Bytes, StandardCharsets.ISO_8859_1);
String recovered = new String(
passthrough.getBytes(StandardCharsets.ISO_8859_1),
StandardCharsets.UTF_8);UTF-8 바이트를 EUC-KR 로 디코딩: ��湲� ← 대체 문자와 오조합이 섞인다
ISO-8859-1 경유 복구: 한글 (원본과 일치: true)바이트가 남아있는 한, 잘못 디코딩한 결과를 다시 원래 바이트로 되돌리고 올바른 인코딩으로 재디코딩하면 복구할 수 있습니다. ISO-8859-1 은 0~255 모든 바이트를 문자 하나에 그대로 대응시키므로 이 왕복에 안전합니다.
// Main.demo3_cp949Only
CharsetEncoder enc = cs.newEncoder()
.onMalformedInput(CodingErrorAction.REPORT)
.onUnmappableCharacter(CodingErrorAction.REPORT);
boolean ok = enc.canEncode("똠");"똠" CP949 인코딩 가능: true
"똠" EUC-KR 인코딩 가능: false"똠"은 KS X 1001 완성형 2350자에 없는 확장 한글이라 EUC-KR 로는 인코딩할 수 없습니다. 레거시 인터페이스가 EUC-KR 이라면 이런 글자가 포함된 값은 저장 전에 걸러내거나 대체해야 합니다.
// Main.demo4_bom
out.write(new byte[]{(byte) 0xEF, (byte) 0xBB, (byte) 0xBF}); // BOM
out.write("이름,나이\n홍길동,30\n".getBytes(StandardCharsets.UTF_8));
String naive = Files.readString(file, StandardCharsets.UTF_8);
String stripped = naive.startsWith("") ? naive.substring(1) : naive;naive 첫 글자 코드: feff
naive.startsWith("이름"): false ← BOM 이 앞에 남아 비교 실패
BOM 제거 후 startsWith("이름"): true// Main.truncateByBytes 핵심부
while (i < s.length()) {
int cp = s.codePointAt(i);
int charCount = Character.charCount(cp);
byte[] piece = s.substring(i, i + charCount).getBytes(cs);
if (out.remaining() < piece.length) break; // 문자 하나를 통째로 포기
out.put(piece);
i += charCount;
}max=10 -> "가나다" (9바이트) ← 4번째 글자(3바이트)가 안 들어가 통째로 제외
max=9 -> "가나다" (9바이트) ← 딱 맞음
max=5 -> "가" (3바이트) ← 2번째 글자도 못 들어감글자 수가 아니라 바이트 수로 잘라야 할 때, 문자 중간에서 끊으면 마지막 글자가 깨집니다. 코드 포인트(서로게이트 쌍 포함) 단위로 통째로 넣을지 말지 결정해야 안전합니다.
// Main.demo6_normalize
String nfd = Normalizer.normalize(nfc, Normalizer.Form.NFD);
boolean same = nfc.equals(nfd);
String backToNfc = Normalizer.normalize(nfd, Normalizer.Form.NFC);NFC 길이: 9, NFD 길이: 18 ← 자모가 분해돼 코드 유닛 수가 늘어난다
NFC.equals(NFD): false
NFD -> NFC 재정규화 후 일치: true// Main.contentDisposition
String encoded = URLEncoder.encode(filename, StandardCharsets.UTF_8)
.replace("+", "%20"); // URLEncoder 는 공백을 +로 바꾼다
return "attachment; filename=\"" + asciiFallback
+ "\"; filename*=UTF-8''" + encoded;attachment; filename="download.xlsx"; filename*=UTF-8''%EB%B3%B4%EA%B3%A0%EC%84%9C%28%EC%B5%9C%EC%A2%85%29.xlsx