문자열을 만드는 다섯 가지 문법, 파싱 로직을 record로 옮기기, 파일명 정규화, 빈 값·null·split의 엣지, 그리고 오토박싱이 오버로딩과 만났을 때의 함정을 다룬다.
같은 문자열을 +, String.join, StringJoiner, 스트림 joining, formatted로 만든다. 열 수가 고정이면 +, 그냥 잇기만 하면 join, 접두·접미가 붙으면 StringJoiner, 변환이 끼면 스트림, 서식이 필요하면 formatted가 각각 가장 짧다.
import java.util.List;
import java.util.StringJoiner;
import java.util.stream.Collectors;
public class Main {
public static void main(String[] args) {
List<String> cols = List.of("O-1", "PAID", "30000");
String a = cols.get(0) + "," + cols.get(1) + "," + cols.get(2); // 열 수 고정
String b = String.join(",", cols); // 잇기만
StringJoiner sj = new StringJoiner(",", "[", "]"); // 접두/접미
for (String c : cols) sj.add(c);
String c = sj.toString();
String d = cols.stream().map(String::toLowerCase).collect(Collectors.joining(",")); // 변환 포함
String e = "%s,%s,%,d".formatted(cols.get(0), cols.get(1), Integer.parseInt(cols.get(2))); // 서식
System.out.println(a); // 출력: O-1,PAID,30000
System.out.println(b); // 출력: O-1,PAID,30000
System.out.println(c); // 출력: [O-1,PAID,30000]
System.out.println(d); // 출력: o-1,paid,30000
System.out.println(e); // 출력: O-1,PAID,30,000
}
}예제 3은 main 안에서 split과 strip을 직접 했다. 파싱을 LogEntry.parse 정적 메서드로 옮기면 한 줄의 형식 지식이 한 곳에 모이고, 집계 코드는 LogEntry의 메서드만 부른다. split("=", 2)의 limit은 값에 =가 또 들어와도 첫 번째에서만 자르게 한다.
import java.util.List;
record LogEntry(String time, String level, String orderId, long amount) {
static LogEntry parse(String line) {
String[] p = line.split("\\|");
return new LogEntry(p[0].strip().substring(11), p[1].strip(), value(p[2]), Long.parseLong(value(p[3])));
}
private static String value(String kv) { return kv.strip().split("=", 2)[1]; } // "order=O-1" → "O-1"
boolean isError() { return level.equals("ERROR"); }
}
public class Main {
public static void main(String[] args) {
String log = """
2026-09-08 10:00:01 | INFO | order=O-1 | amount=30000
2026-09-08 10:00:05 | ERROR | order=O-2 | amount=15000
2026-09-08 10:00:09 | INFO | order=O-3 | amount=99000
""";
List<LogEntry> entries = log.lines().map(LogEntry::parse).toList();
entries.forEach(System.out::println);
// 출력:
// LogEntry[time=10:00:01, level=INFO, orderId=O-1, amount=30000]
// LogEntry[time=10:00:05, level=ERROR, orderId=O-2, amount=15000]
// LogEntry[time=10:00:09, level=INFO, orderId=O-3, amount=99000]
long errors = entries.stream().filter(LogEntry::isError).count();
long total = entries.stream().filter(e -> !e.isError()).mapToLong(LogEntry::amount).sum();
System.out.println("ERROR " + errors + "건, 정상 합계 %,d원".formatted(total));
// 출력: ERROR 1건, 정상 합계 129,000원
}
}업로드된 파일명을 URL·저장소에 안전한 형태로 바꾼다. 확장자는 보존하고, 본문에서 허용 문자(영문·숫자·한글) 외의 연속 구간은 - 하나로 바꾸며, 앞뒤의 -는 제거한다. replaceAll은 정규식이므로 [^...](부정 문자 클래스)와 ^, $(앵커)를 그대로 쓸 수 있다.
import java.util.List;
public class Main {
static String normalize(String raw) {
String name = raw.strip().toLowerCase();
int dot = name.lastIndexOf('.');
String base = dot < 0 ? name : name.substring(0, dot);
String ext = dot < 0 ? "" : name.substring(dot);
base = base.replaceAll("[^a-z0-9가-힣]+", "-") // 허용 외 문자 구간 → -
.replaceAll("^-|-$", ""); // 앞뒤 - 제거
return base + ext;
}
public static void main(String[] args) {
List<String> raw = List.of(" My Report (Final).PDF ", "사진 2026.09.08.jpg", "___temp___", "invoice#2026 v2.xlsx");
for (String r : raw) {
System.out.println("[" + r + "] → " + normalize(r));
}
// 출력:
// [ My Report (Final).PDF ] → my-report-final.pdf
// [사진 2026.09.08.jpg] → 사진-2026-09-08.jpg
// [___temp___] → temp
// [invoice#2026 v2.xlsx] → invoice-2026-v2.xlsx
}
}isEmpty와 isBlank의 차이, null과의 연결·비교, split이 빈 문자열과 끝의 빈 값을 다루는 방식, substring의 허용 범위, strip과 trim의 유니코드 공백 처리 차이를 한 번에 본다. 이 중 split의 끝 빈 값 제거는 CSV 파싱에서 열 수가 달라지는 실제 버그의 원인이다.
import java.util.Arrays;
import java.util.Objects;
public class Main {
public static void main(String[] args) {
String empty = "", blank = " ", none = null;
System.out.println(empty.isEmpty() + " " + blank.isEmpty() + " " + blank.isBlank()); // 출력: true false true
System.out.println(Objects.requireNonNullElse(none, "기본값")); // 출력: 기본값
System.out.println("null".equals(none) + " " + (none + "")); // 출력: false null
System.out.println(Arrays.toString("".split(",")) + " " + "".split(",").length); // 출력: [] 1 (빈 문자열 하나)
System.out.println(Arrays.toString("a,,b,,".split(","))); // 출력: [a, , b] (끝의 빈 값은 버림)
System.out.println(Arrays.toString("a,,b,,".split(",", -1))); // 출력: [a, , b, , ] (limit -1: 유지)
System.out.println("abc".substring(3).isEmpty()); // 출력: true (길이와 같은 인덱스는 허용)
try {
"abc".substring(4);
} catch (StringIndexOutOfBoundsException e) {
System.out.println("범위 초과: " + e.getMessage()); // 출력: 범위 초과: Range [4, 3) out of bounds for length 3
}
String emSpace = " x"; // 유니코드 공백(EM SPACE)
System.out.println(emSpace.strip().length() + " " + emSpace.trim().length()); // 출력: 1 2 (trim은 ASCII 공백만)
}
}List<Integer>에는 remove(int index)와 remove(Object o)가 둘 다 있다. int를 넘기면 박싱 없이 인덱스 버전이 먼저 선택된다. 값을 지우려면 Integer.valueOf로 명시적으로 박싱해야 한다. 같은 원리로 Set<Long>에 int로 contains하면 Integer로 박싱되어 절대 찾지 못한다.
import java.util.*;
public class Main {
public static void main(String[] args) {
List<Integer> ids = new ArrayList<>(List.of(10, 20, 30, 40));
ids.remove(1); // int → remove(int index): 인덱스 1의 20 삭제
System.out.println(ids); // 출력: [10, 30, 40]
ids.remove(Integer.valueOf(30)); // Integer → remove(Object): 값 30 삭제
System.out.println(ids); // 출력: [10, 40]
ids.remove((Integer) 40); // 캐스팅으로도 값 삭제
System.out.println(ids); // 출력: [10]
try {
ids.remove(10); // 값 10을 지우려 했지만 인덱스 10
} catch (IndexOutOfBoundsException e) {
System.out.println("인덱스 10 없음"); // 출력: 인덱스 10 없음
}
List<Integer> big = List.of(1000);
Integer x = big.get(0), y = 1000;
System.out.println(big.get(0) == 1000); // 출력: true (int와 비교 → 언박싱)
System.out.println(x == y); // 출력: false (Integer끼리 == → 주소, 캐시 범위 밖)
System.out.println(x.equals(y)); // 출력: true
Set<Long> longs = new HashSet<>(List.of(5L));
System.out.println(longs.contains(5) + " " + longs.contains(5L)); // 출력: false true (Integer(5)는 Long(5)와 다름)
}
}