ItemReader<String> 을 구현하는 RangeReader 를 만드세요. 생성자로 from, to 를 받아 "item-from" 부터 "item-to" 까지 순서대로 반환하고, 끝나면 null 을 반환합니다. read() 가 몇 번 호출되었는지 세는 callCount() 도 추가하세요.
public class RangeReader implements ItemReader<String> {
private final int to;
private int current;
private int calls;
public RangeReader(int from, int to) { this.current = from; this.to = to; }
@Override
public String read() {
calls++;
if (current > to) return null;
return "item-" + current++;
}
public int callCount() { return calls; }
}
// 테스트
RangeReader r = new RangeReader(1, 3);
System.out.println(r.read() + " " + r.read() + " " + r.read() + " " + r.read());
System.out.println("calls=" + r.callCount());
// 출력:
// item-1 item-2 item-3 null
// calls=4Step 을 이용해 1~1000 의 정수 중 3의 배수만 골라 제곱한 값을 리스트에 모으는 배치를 작성하세요. chunk 크기는 50 으로 하고, StepExecution 의 readCount, filterCount, writeCount, commitCount 를 출력하세요. 각 값이 왜 그렇게 나오는지 주석으로 설명하세요.
List<Integer> numbers = IntStream.rangeClosed(1, 1000).boxed().toList();
List<Long> squares = new ArrayList<>();
Step<Integer, Long> step = new Step<>("squareMultiplesOf3",
ItemReader.of(numbers),
n -> n % 3 == 0 ? (long) n * n : null, // 3의 배수가 아니면 필터
squares::addAll,
50);
StepExecution exec = step.execute();
System.out.println(exec.summary());
// 출력: [squareMultiplesOf3] COMPLETED read=1000 filtered=667 written=333 commits=20 2ms
//
// read=1000 : 전부 읽음
// filtered=667 : 1000 - 333 (1~1000 중 3의 배수는 333개)
// written=333 : 필터 통과분
// commits=20 : 청크는 "reader.read() 를 chunkSize(50)번 호출" 단위로 끊긴다.
// 1000 / 50 = 20 청크. 각 청크에는 3의 배수가 16~17개뿐이지만
// 청크가 비지 않는 한 write 되므로 20번 커밋된다.이 문제의 교훈: 청크 크기는 "읽은 건수" 기준입니다(이 구현 및 Spring Batch 모두). 필터가 많으면 실제 write 되는 건수는 chunkSize 보다 훨씬 작을 수 있습니다. 쓰기 효율을 위해 "쓰기 건수 기준 청크"를 원한다면 Step 을 수정해야 합니다(문제 3).
문제 2의 교훈을 바탕으로, Step 을 복사해 필터 통과 건수(chunk.size())가 chunkSize 에 도달할 때 쓰기를 수행하는 WriteSizedStep 을 만드세요. 같은 입력(1~1000, 3의 배수, chunk=50)에서 commitCount 가 몇이 되는지 확인하세요.
public class WriteSizedStep<I, O> {
// 필드/생성자는 Step 과 동일 (생략)
public StepExecution execute() {
LocalDateTime start = LocalDateTime.now();
long read = 0, filtered = 0, written = 0, commits = 0;
try {
reader.open();
List<O> chunk = new ArrayList<>(chunkSize);
I item;
while ((item = reader.read()) != null) {
read++;
O out = processor.process(item);
if (out == null) { filtered++; continue; }
chunk.add(out);
if (chunk.size() == chunkSize) { // 쓰기 건수 기준
writer.write(chunk);
written += chunk.size();
commits++;
chunk.clear();
}
}
if (!chunk.isEmpty()) { // 남은 꼬리
writer.write(chunk);
written += chunk.size();
commits++;
}
return new StepExecution(name, StepExecution.Status.COMPLETED, start, LocalDateTime.now(),
read, filtered, written, commits, null);
} catch (Exception e) {
return new StepExecution(name, StepExecution.Status.FAILED, start, LocalDateTime.now(),
read, filtered, written, commits, e.toString());
} finally {
try { reader.close(); } catch (Exception ignored) {}
}
}
}
// 출력: [squareMultiplesOf3] COMPLETED read=1000 filtered=667 written=333 commits=7 2ms
// 333 / 50 = 6 청크 + 나머지 33건 1 청크 = 7 커밋
//
// 트레이드오프: 쓰기 효율은 좋아지지만, 한 청크가 "읽기 기준" 으로는 150건에 걸치므로
// 실패 시 롤백/재처리 범위가 커진다. Spring Batch 가 읽기 기준을 택한 이유다.