java-src/batch/02_file_io/ 에서 javac *.java && java Main 100000 으로 실행합니다. data/ 폴더에 샘플 파일을 만들고 벤치마크와 예제를 순서대로 실행합니다.
같은 CSV 를 다섯 가지 방식으로 읽어 줄 수를 셉니다. 각 방식의 시스템 콜/메모리 특성을 시간으로 확인합니다.
static long readUnbuffered(Path p) throws IOException { // 글자 단위
long lines = 0;
try (FileReader r = new FileReader(p.toFile(), StandardCharsets.UTF_8)) {
int c;
while ((c = r.read()) != -1) if (c == '\n') lines++;
}
return lines;
}
static long readBuffered(Path p) throws IOException { // 줄 단위, 8KB 버퍼
long lines = 0;
try (BufferedReader r = Files.newBufferedReader(p, StandardCharsets.UTF_8)) {
while (r.readLine() != null) lines++;
}
return lines;
}
static long readAllLines(Path p) throws IOException { // 전체 로딩
return Files.readAllLines(p, StandardCharsets.UTF_8).size();
}
static long readLinesStream(Path p) throws IOException { // 지연 스트림
try (Stream<String> s = Files.lines(p, StandardCharsets.UTF_8)) {
return s.count();
}
}
static long readMapped(Path p) throws IOException { // mmap 바이트 스캔
long lines = 0;
try (FileChannel ch = FileChannel.open(p, StandardOpenOption.READ)) {
long size = ch.size(), pos = 0;
while (pos < size) {
long len = Math.min(Integer.MAX_VALUE, size - pos); // 2GB 단위로 나눠 매핑
MappedByteBuffer buf = ch.map(FileChannel.MapMode.READ_ONLY, pos, len);
for (int i = 0; i < len; i++) if (buf.get(i) == '\n') lines++;
pos += len;
}
}
return lines;
}
// 출력 (100,000행, 약 4.4MB, 환경에 따라 다름):
// FileReader.read() 한 글자씩 100,001 lines 92 ms
// BufferedReader.readLine() 100,001 lines 24 ms
// Files.readAllLines (전체 로딩) 100,001 lines 36 ms
// Files.lines (스트림) 100,001 lines 22 ms
// MappedByteBuffer 바이트 스캔 100,001 lines 5 msreadAllLines 는 시간도 느리지 않아 보이지만, 힙에 100,001개의 String 이 동시에 살아 있습니다. 100만 행이면 수백 MB 입니다(03 레슨에서 측정).
inQuotes 상태 하나로 동작하는 상태 머신입니다.
public static List<String> parseLine(String line) {
List<String> fields = new ArrayList<>();
StringBuilder cur = new StringBuilder();
boolean inQuotes = false;
for (int i = 0; i < line.length(); i++) {
char c = line.charAt(i);
if (inQuotes) {
if (c == '"') {
if (i + 1 < line.length() && line.charAt(i + 1) == '"') {
cur.append('"'); i++; // "" → "
} else {
inQuotes = false; // 닫는 따옴표
}
} else cur.append(c);
} else {
if (c == '"') inQuotes = true;
else if (c == ',') { fields.add(cur.toString()); cur.setLength(0); }
else cur.append(c);
}
}
fields.add(cur.toString()); // 마지막 필드
return fields;
}
public static String quote(String v) { // 쓰기용: 필요할 때만 감싼다
if (v.indexOf(',') < 0 && v.indexOf('"') < 0 && v.indexOf('\n') < 0) return v;
return "\"" + v.replace("\"", "\"\"") + "\"";
}
// CsvParser.parseLine("1,2026-09-01,1001,WITHDRAW,5000,\"lunch, with team\"")
// 출력: [1, 2026-09-01, 1001, WITHDRAW, 5000, lunch, with team] ← 6개 필드
// CsvParser.parseLine("2,2026-09-01,1001,DEPOSIT,5000,\"book, \"\"Java\"\" 21\"")
// 출력: [2, 2026-09-01, 1001, DEPOSIT, 5000, book, "Java" 21]
// "1,...".split(",") 이었다면 → 7개 필드, amount 다음에 "lunch 가 들어감100만 행이어도 힙에는 한 줄과 TreeMap<계좌, 잔액> 500개만 있습니다.
record Transaction(long id, String date, int accountId, String type, long amount, String memo) {
static Transaction fromCsv(String line) {
List<String> f = CsvParser.parseLine(line);
return new Transaction(Long.parseLong(f.get(0)), f.get(1), Integer.parseInt(f.get(2)),
f.get(3), Long.parseLong(f.get(4)), f.get(5));
}
static Transaction fromFixed(String line) { // id(10) date(8) acct(6) type(1) amount(12)
return new Transaction(
Long.parseLong(line.substring(0, 10).trim()),
line.substring(10, 18),
Integer.parseInt(line.substring(18, 24).trim()),
line.charAt(24) == 'W' ? "WITHDRAW" : "DEPOSIT",
Long.parseLong(line.substring(25, 37).trim()), "");
}
long signedAmount() { return type.equals("DEPOSIT") ? amount : -amount; }
}
Map<Integer, Long> balanceByAccount = new TreeMap<>();
try (BufferedReader r = Files.newBufferedReader(csv, StandardCharsets.UTF_8)) {
r.readLine(); // 헤더
String line;
while ((line = r.readLine()) != null) {
Transaction t = Transaction.fromCsv(line);
balanceByAccount.merge(t.accountId(), t.signedAmount(), Long::sum);
}
}
// 출력: CSV 집계: 100,000건 파싱, 계좌 500개
// 계좌 1000 잔액 변동 -1,234,000원 ...
long withdraws;
try (Stream<String> lines = Files.lines(fixed, StandardCharsets.US_ASCII)) {
withdraws = lines.map(Transaction::fromFixed).filter(t -> t.type().equals("WITHDRAW")).count();
}
// 출력: 고정길이 파싱: 출금 건수 60,1xx / 100000헤더를 유지하며 25,000행씩 자르고, 다시 하나로 합칩니다. 원본과 크기가 같으면 손실이 없습니다.
static List<Path> split(Path src, Path outDir, int linesPerFile) throws IOException {
Files.createDirectories(outDir);
List<Path> parts = new ArrayList<>();
try (BufferedReader r = Files.newBufferedReader(src, StandardCharsets.UTF_8)) {
String header = r.readLine();
String line;
BufferedWriter w = null;
int count = 0;
try {
while ((line = r.readLine()) != null) {
if (w == null || count == linesPerFile) { // 새 파트 시작
if (w != null) w.close();
Path part = outDir.resolve(String.format("part-%03d.csv", parts.size()));
parts.add(part);
w = Files.newBufferedWriter(part, StandardCharsets.UTF_8);
w.write(header); w.newLine();
count = 0;
}
w.write(line); w.newLine();
count++;
}
} finally {
if (w != null) w.close();
}
}
return parts;
}
static Path merge(List<Path> parts, Path dest) throws IOException {
try (BufferedWriter w = Files.newBufferedWriter(dest, StandardCharsets.UTF_8)) {
boolean first = true;
for (Path part : parts) {
try (BufferedReader r = Files.newBufferedReader(part, StandardCharsets.UTF_8)) {
String header = r.readLine();
if (first) { w.write(header); w.newLine(); first = false; } // 헤더는 한 번만
String line;
while ((line = r.readLine()) != null) { w.write(line); w.newLine(); }
}
}
}
return dest;
}
// 출력:
// 분할: 4개 파일 -> [data\parts\part-000.csv, data\parts\part-001.csv, data\parts\part-002.csv, data\parts\part-003.csv]
// 병합: data\merged.csv (4,412,345 bytes, 원본 4,412,345 bytes)static Path writeReportAtomically(Path dest, Map<Integer, Long> balances) throws IOException {
Path tmp = dest.resolveSibling(dest.getFileName() + ".tmp"); // 같은 디렉터리!
try (BufferedWriter w = Files.newBufferedWriter(tmp, StandardCharsets.UTF_8)) {
w.write("accountId,balanceChange"); w.newLine();
for (Map.Entry<Integer, Long> e : balances.entrySet()) {
w.write(e.getKey() + "," + e.getValue()); w.newLine();
}
} // 여기까지 죽으면 .tmp 만 남음
Files.move(tmp, dest, StandardCopyOption.ATOMIC_MOVE, StandardCopyOption.REPLACE_EXISTING);
return dest;
}
// 출력: 리포트 작성 완료: data\balance_report.csv (6234 bytes)
// 쓰는 도중 강제 종료 → data\balance_report.csv.tmp 만 존재, 소비자는 정식 파일이 없으므로 대기