chore : deploy 수정

This commit is contained in:
2026-08-11 17:55:45 +09:00
parent a7077fc693
commit fd26a086f7
41 changed files with 23249 additions and 3 deletions
@@ -2,8 +2,10 @@ package company.specialsource;
import org.springframework.boot.SpringApplication;
import org.springframework.boot.autoconfigure.SpringBootApplication;
import org.springframework.scheduling.annotation.EnableScheduling;
@SpringBootApplication
@EnableScheduling
public class SpecialsourceHomepageApplication {
public static void main(String[] args) {
@@ -21,7 +21,7 @@ public class DataInitializer implements CommandLineRunner {
if (userRepository.findByUsername("admin").isEmpty()) {
userRepository.save(User.builder()
.username("admin")
.password(passwordEncoder.encode("1234"))
.password(passwordEncoder.encode("Ppasse444$"))
.role("ROLE_USER")
.build());
log.info("초기 관리자 계정(admin)이 생성되었습니다.");
@@ -0,0 +1,46 @@
package company.specialsource.config;
import jakarta.servlet.FilterChain;
import jakarta.servlet.ServletException;
import jakarta.servlet.http.HttpServletRequest;
import jakarta.servlet.http.HttpServletResponse;
import lombok.RequiredArgsConstructor;
import org.springframework.security.authentication.UsernamePasswordAuthenticationToken;
import org.springframework.security.core.context.SecurityContextHolder;
import org.springframework.stereotype.Component;
import org.springframework.web.filter.OncePerRequestFilter;
import java.io.IOException;
import java.util.List;
@Component
@RequiredArgsConstructor
public class JwtAuthenticationFilter extends OncePerRequestFilter {
private static final String HEADER_PREFIX = "Bearer ";
private final JwtTokenProvider jwtTokenProvider;
@Override
protected void doFilterInternal(
HttpServletRequest request,
HttpServletResponse response,
FilterChain filterChain
) throws ServletException, IOException {
String header = request.getHeader("Authorization");
if (header != null && header.startsWith(HEADER_PREFIX)) {
String token = header.substring(HEADER_PREFIX.length());
if (jwtTokenProvider.validateToken(token)) {
String username = jwtTokenProvider.getUsername(token);
UsernamePasswordAuthenticationToken authentication =
new UsernamePasswordAuthenticationToken(username, null, List.of());
SecurityContextHolder.getContext().setAuthentication(authentication);
}
}
filterChain.doFilter(request, response);
}
}
@@ -1,25 +1,33 @@
package company.specialsource.config;
import lombok.RequiredArgsConstructor;
import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Configuration;
import org.springframework.security.config.annotation.web.builders.HttpSecurity;
import org.springframework.security.config.annotation.web.configuration.EnableWebSecurity;
import org.springframework.security.config.http.SessionCreationPolicy;
import org.springframework.security.crypto.bcrypt.BCryptPasswordEncoder;
import org.springframework.security.crypto.password.PasswordEncoder;
import org.springframework.security.web.SecurityFilterChain;
import org.springframework.security.web.authentication.UsernamePasswordAuthenticationFilter;
@Configuration
@EnableWebSecurity
@RequiredArgsConstructor
public class SecurityConfig {
private final JwtAuthenticationFilter jwtAuthenticationFilter;
@Bean
public SecurityFilterChain filterChain(HttpSecurity http) throws Exception {
http
.csrf(csrf -> csrf.disable()) // REST API 통신을 위해 CSRF 비활성화
.sessionManagement(session -> session.sessionCreationPolicy(SessionCreationPolicy.STATELESS))
.authorizeHttpRequests(auth -> auth
.requestMatchers("/api/login").permitAll() // 로그인 API는 누구나 접근 가능
.anyRequest().authenticated()
);
)
.addFilterBefore(jwtAuthenticationFilter, UsernamePasswordAuthenticationFilter.class);
return http.build();
}
@@ -0,0 +1,56 @@
package company.specialsource.controller;
import company.specialsource.scraper.BoardSource;
import company.specialsource.scraper.BoardSummaryResponse;
import company.specialsource.scraper.ScrapedPostQueryService;
import company.specialsource.scraper.ScrapedPostResponse;
import lombok.RequiredArgsConstructor;
import org.springframework.data.domain.Page;
import org.springframework.data.domain.PageRequest;
import org.springframework.data.domain.Pageable;
import org.springframework.http.ResponseEntity;
import org.springframework.web.bind.annotation.*;
import java.util.Arrays;
import java.util.List;
@RestController
@RequestMapping("/api/scraped-posts")
@RequiredArgsConstructor
public class ScrapedPostController {
private final ScrapedPostQueryService scrapedPostQueryService;
@GetMapping
public ResponseEntity<?> getPosts(
@RequestParam(required = false) String board,
@RequestParam(defaultValue = "0") int page,
@RequestParam(defaultValue = "20") int size
) {
if (board != null && !isValidBoardKey(board)) {
return ResponseEntity.badRequest().body("존재하지 않는 게시판 키입니다: " + board);
}
Pageable pageable = PageRequest.of(page, Math.min(size, 100));
Page<ScrapedPostResponse> posts = scrapedPostQueryService.getPosts(board, pageable);
return ResponseEntity.ok(posts);
}
@GetMapping("/boards")
public ResponseEntity<List<BoardSummaryResponse>> getBoards() {
List<BoardSummaryResponse> boards = Arrays.stream(BoardSource.values())
.map(BoardSummaryResponse::from)
.toList();
return ResponseEntity.ok(boards);
}
@PostMapping("/{id}/view")
public ResponseEntity<Void> markViewed(@PathVariable Long id) {
scrapedPostQueryService.markViewed(id);
return ResponseEntity.noContent().build();
}
private boolean isValidBoardKey(String boardKey) {
return Arrays.stream(BoardSource.values()).anyMatch(source -> source.name().equals(boardKey));
}
}
@@ -0,0 +1,56 @@
package company.specialsource.entity;
import jakarta.persistence.*;
import lombok.*;
import java.time.LocalDateTime;
@Entity
@Table(name = "scraped_posts",
uniqueConstraints = @UniqueConstraint(columnNames = {"board_key", "link"}))
@Getter
@NoArgsConstructor(access = AccessLevel.PROTECTED)
@AllArgsConstructor
@Builder
public class ScrapedPost {
@Id
@GeneratedValue(strategy = GenerationType.IDENTITY)
private Long id;
@Column(name = "board_key", nullable = false)
private String boardKey; // BoardSource enum name, 예: RULIWEB_BEST
@Column(name = "site_name", nullable = false)
private String siteName; // 예: 루리웹, 디씨인사이드
@Column(name = "board_name", nullable = false)
private String boardName; // 예: 실시간베스트, 주짓수 갤러리
@Column(nullable = false, length = 500)
private String title;
@Column(nullable = false, length = 1000)
private String link;
private String author;
private String postedAt; // 사이트마다 표기 형식이 달라(HH:mm, yy.MM.dd 등) 원문 그대로 저장
private Integer viewCount;
private Integer likeCount;
@Column(columnDefinition = "TEXT")
private String content; // 게시물 상세페이지에서 추출한 본문 (정제된 HTML, 원본 문단/링크/이미지 위치 보존)
@Column(name = "image_urls", columnDefinition = "TEXT")
private String imageUrls; // 본문에 포함된 이미지 URL, 줄바꿈으로 구분
@Column(name = "scraped_at", nullable = false)
private LocalDateTime scrapedAt;
@Setter
@Column(name = "viewed_at")
private LocalDateTime viewedAt; // 게시판 조회 시각. null이면 아직 안 읽은 글
}
@@ -0,0 +1,20 @@
package company.specialsource.repository;
import company.specialsource.entity.ScrapedPost;
import org.springframework.data.domain.Page;
import org.springframework.data.domain.Pageable;
import org.springframework.data.jpa.repository.JpaRepository;
import org.springframework.transaction.annotation.Transactional;
import java.time.LocalDateTime;
public interface ScrapedPostRepository extends JpaRepository<ScrapedPost, Long> {
boolean existsByBoardKeyAndLink(String boardKey, String link);
Page<ScrapedPost> findByBoardKeyAndViewedAtIsNullOrderByScrapedAtDesc(String boardKey, Pageable pageable);
Page<ScrapedPost> findByViewedAtIsNullOrderByScrapedAtDesc(Pageable pageable);
@Transactional
long deleteByScrapedAtBefore(LocalDateTime cutoff);
}
@@ -0,0 +1,9 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import java.util.List;
public interface BoardParser {
List<ScrapedPostDto> parse(Document document);
}
@@ -0,0 +1,37 @@
package company.specialsource.scraper;
import lombok.RequiredArgsConstructor;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.springframework.stereotype.Service;
import java.io.IOException;
import java.util.List;
@Service
@RequiredArgsConstructor
public class BoardScraperService {
private static final String USER_AGENT =
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36";
private static final int TIMEOUT_MS = 10_000;
private final RuliwebBoardListParser ruliwebBoardListParser;
private final RuliwebMypiParser ruliwebMypiParser;
private final DcGalleryParser dcGalleryParser;
public List<ScrapedPostDto> scrape(BoardSource source) throws IOException {
Document document = Jsoup.connect(source.getUrl())
.userAgent(USER_AGENT)
.timeout(TIMEOUT_MS)
.get();
BoardParser parser = switch (source.getParserType()) {
case RULIWEB_BOARD_LIST -> ruliwebBoardListParser;
case RULIWEB_MYPI_BEST -> ruliwebMypiParser;
case DC_GALLERY_LIST -> dcGalleryParser;
};
return parser.parse(document);
}
}
@@ -0,0 +1,27 @@
package company.specialsource.scraper;
import lombok.Getter;
@Getter
public enum BoardSource {
RULIWEB_BEST("루리웹", "실시간베스트", "https://bbs.ruliweb.com/best/selection", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
RULIWEB_HUMOR("루리웹", "유머", "https://bbs.ruliweb.com/best/humor", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
RULIWEB_MYPI("루리웹", "마이피", "https://mypi.ruliweb.com/", ParserType.RULIWEB_MYPI_BEST, DetailParserType.RULIWEB),
RULIWEB_HOTDEAL("루리웹", "핫딜", "https://bbs.ruliweb.com/market/board/1020", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
DC_BJJ("디씨인사이드", "주짓수 갤러리", "https://gall.dcinside.com/mgallery/board/lists/?id=bjj", ParserType.DC_GALLERY_LIST, DetailParserType.DC),
DC_SINGULARITY("디씨인사이드", "특이점이 온다", "https://gall.dcinside.com/mgallery/board/lists?id=thesingularity", ParserType.DC_GALLERY_LIST, DetailParserType.DC);
private final String siteName;
private final String boardName;
private final String url;
private final ParserType parserType;
private final DetailParserType detailParserType;
BoardSource(String siteName, String boardName, String url, ParserType parserType, DetailParserType detailParserType) {
this.siteName = siteName;
this.boardName = boardName;
this.url = url;
this.parserType = parserType;
this.detailParserType = detailParserType;
}
}
@@ -0,0 +1,7 @@
package company.specialsource.scraper;
public record BoardSummaryResponse(String boardKey, String siteName, String boardName) {
public static BoardSummaryResponse from(BoardSource source) {
return new BoardSummaryResponse(source.name(), source.getSiteName(), source.getBoardName());
}
}
@@ -0,0 +1,53 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import org.jsoup.nodes.Element;
import org.springframework.stereotype.Component;
import java.util.ArrayList;
import java.util.List;
/**
* 디씨인사이드 갤러리 목록 파서. 공지/설문 행은 제외하고 일반 게시글만 수집한다.
*/
@Component
public class DcGalleryParser implements BoardParser {
@Override
public List<ScrapedPostDto> parse(Document document) {
List<ScrapedPostDto> posts = new ArrayList<>();
for (Element row : document.select("tr.ub-content")) {
if (!row.hasClass("us-post") || "icon_notice".equals(row.attr("data-type"))) {
continue;
}
Element titleAnchor = row.selectFirst("td.gall_tit a");
if (titleAnchor == null) {
continue;
}
String title = titleAnchor.text().trim();
String link = titleAnchor.attr("abs:href");
if (title.isEmpty() || link.isEmpty()) {
continue;
}
Element writerEl = row.selectFirst("td.gall_writer");
String author = writerEl == null ? "" : writerEl.attr("data-nick");
if (author.isBlank() && writerEl != null) {
author = writerEl.text().trim();
}
Element dateEl = row.selectFirst("td.gall_date");
String postedAt = dateEl == null ? "" : dateEl.hasAttr("title") ? dateEl.attr("title") : dateEl.text().trim();
Integer viewCount = ScraperTextUtils.parseIntSafe(row.select("td.gall_count").text());
Integer likeCount = ScraperTextUtils.parseIntSafe(row.select("td.gall_recommend").text());
posts.add(new ScrapedPostDto(title, link, author, postedAt, viewCount, likeCount));
}
return posts;
}
}
@@ -0,0 +1,13 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import org.springframework.stereotype.Component;
@Component
public class DcPostDetailParser implements DetailParser {
@Override
public PostDetail parse(Document document) {
return ScraperTextUtils.extractDetail(document.selectFirst("div.write_div"));
}
}
@@ -0,0 +1,7 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
public interface DetailParser {
PostDetail parse(Document document);
}
@@ -0,0 +1,6 @@
package company.specialsource.scraper;
public enum DetailParserType {
RULIWEB,
DC
}
@@ -0,0 +1,7 @@
package company.specialsource.scraper;
public enum ParserType {
RULIWEB_BOARD_LIST, // 표 형태 게시판 (실시간베스트, 유머, 핫딜)
RULIWEB_MYPI_BEST, // 마이피 실시간 인기글 위젯
DC_GALLERY_LIST // 디씨인사이드 갤러리 게시판
}
@@ -0,0 +1,10 @@
package company.specialsource.scraper;
import java.util.List;
/**
* content는 원본 게시물의 문단/굵게/링크/이미지 위치를 보존한 정제된(sanitized) HTML이다.
*/
public record PostDetail(String content, List<String> imageUrls) {
static final PostDetail EMPTY = new PostDetail("", List.of());
}
@@ -0,0 +1,34 @@
package company.specialsource.scraper;
import lombok.RequiredArgsConstructor;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.springframework.stereotype.Service;
import java.io.IOException;
@Service
@RequiredArgsConstructor
public class PostDetailScraperService {
private static final String USER_AGENT =
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36";
private static final int TIMEOUT_MS = 10_000;
private final RuliwebPostDetailParser ruliwebPostDetailParser;
private final DcPostDetailParser dcPostDetailParser;
public PostDetail scrape(String url, DetailParserType type) throws IOException {
Document document = Jsoup.connect(url)
.userAgent(USER_AGENT)
.timeout(TIMEOUT_MS)
.get();
DetailParser parser = switch (type) {
case RULIWEB -> ruliwebPostDetailParser;
case DC -> dcPostDetailParser;
};
return parser.parse(document);
}
}
@@ -0,0 +1,48 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import org.jsoup.nodes.Element;
import org.springframework.stereotype.Component;
import java.util.ArrayList;
import java.util.List;
/**
* 루리웹 표 형태 게시판(실시간베스트, 유머, 핫딜) 공통 파서.
* 게시판마다 tr.table_body 구조는 같고, 공지(class=notice) 행만 다르게 섞여 있어 걸러낸다.
*/
@Component
public class RuliwebBoardListParser implements BoardParser {
@Override
public List<ScrapedPostDto> parse(Document document) {
List<ScrapedPostDto> posts = new ArrayList<>();
for (Element row : document.select("tr.table_body")) {
if (row.hasClass("notice")) {
continue;
}
Element titleAnchor = row.selectFirst("td.subject a");
if (titleAnchor == null) {
continue;
}
Element strong = titleAnchor.selectFirst("strong");
String title = (strong != null ? strong.text() : titleAnchor.text()).trim();
String link = titleAnchor.attr("abs:href");
if (title.isEmpty() || link.isEmpty()) {
continue;
}
String author = row.select("td.writer").text().trim();
String postedAt = row.select("td.time").text().trim();
Integer likeCount = ScraperTextUtils.parseIntSafe(row.select("td.recomd").text());
Integer viewCount = ScraperTextUtils.parseIntSafe(row.select("td.hit").text());
posts.add(new ScrapedPostDto(title, link, author, postedAt, viewCount, likeCount));
}
return posts;
}
}
@@ -0,0 +1,36 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import org.jsoup.nodes.Element;
import org.springframework.stereotype.Component;
import java.util.ArrayList;
import java.util.List;
/**
* 마이피(mypi.ruliweb.com)의 "실시간 인기글" 위젯 파서.
* 작성자/날짜/조회수는 목록에 노출되지 않아 title, link만 수집한다.
*/
@Component
public class RuliwebMypiParser implements BoardParser {
@Override
public List<ScrapedPostDto> parse(Document document) {
List<ScrapedPostDto> posts = new ArrayList<>();
for (Element anchor : document.select("ul.right_best_list li.right_best_list_item a.txt_link")) {
String title = anchor.text().trim();
String link = anchor.attr("abs:href");
if (title.isEmpty() || link.isEmpty()) {
continue;
}
// 위젯 목록에 광고(외부 리다이렉트 링크)가 섞여 들어오는 경우가 있어 루리웹 도메인만 통과시킨다.
if (!link.contains("ruliweb.com")) {
continue;
}
posts.add(new ScrapedPostDto(title, link, null, null, null, null));
}
return posts;
}
}
@@ -0,0 +1,13 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import org.springframework.stereotype.Component;
@Component
public class RuliwebPostDetailParser implements DetailParser {
@Override
public PostDetail parse(Document document) {
return ScraperTextUtils.extractDetail(document.selectFirst("div.view_content"));
}
}
@@ -0,0 +1,37 @@
package company.specialsource.scraper;
import company.specialsource.repository.ScrapedPostRepository;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.scheduling.annotation.Scheduled;
import org.springframework.stereotype.Component;
import java.time.LocalDateTime;
/**
* 오래된 스크랩 게시물을 주기적으로 정리한다.
* 스크래핑으로 계속 쌓이기만 하는 캐시성 데이터라 일정 기간이 지나면 삭제해도 무방하다.
*/
@Slf4j
@Component
@RequiredArgsConstructor
public class ScrapedPostCleanupScheduler {
private final ScrapedPostRepository scrapedPostRepository;
@Value("${scraping.retention-days:7}")
private int retentionDays;
@Scheduled(
initialDelayString = "${scraping.cleanup-initial-delay-ms:60000}",
fixedDelayString = "${scraping.cleanup-interval-ms:86400000}"
)
public void deleteExpiredPosts() {
LocalDateTime cutoff = LocalDateTime.now().minusDays(retentionDays);
long deletedCount = scrapedPostRepository.deleteByScrapedAtBefore(cutoff);
if (deletedCount > 0) {
log.info("[정리] 스크랩 {}일 경과 게시물 {}건 삭제 (기준: {})", retentionDays, deletedCount, cutoff);
}
}
}
@@ -0,0 +1,11 @@
package company.specialsource.scraper;
public record ScrapedPostDto(
String title,
String link,
String author,
String postedAt,
Integer viewCount,
Integer likeCount
) {
}
@@ -0,0 +1,35 @@
package company.specialsource.scraper;
import company.specialsource.entity.ScrapedPost;
import company.specialsource.repository.ScrapedPostRepository;
import jakarta.persistence.EntityNotFoundException;
import lombok.RequiredArgsConstructor;
import org.springframework.data.domain.Page;
import org.springframework.data.domain.Pageable;
import org.springframework.stereotype.Service;
import org.springframework.transaction.annotation.Transactional;
import java.time.LocalDateTime;
@Service
@RequiredArgsConstructor
public class ScrapedPostQueryService {
private final ScrapedPostRepository scrapedPostRepository;
public Page<ScrapedPostResponse> getPosts(String boardKey, Pageable pageable) {
// 이미 조회한(viewedAt != null) 글은 목록에서 제외한다.
Page<ScrapedPost> page = (boardKey == null || boardKey.isBlank())
? scrapedPostRepository.findByViewedAtIsNullOrderByScrapedAtDesc(pageable)
: scrapedPostRepository.findByBoardKeyAndViewedAtIsNullOrderByScrapedAtDesc(boardKey, pageable);
return page.map(ScrapedPostResponse::from);
}
@Transactional
public void markViewed(Long id) {
ScrapedPost post = scrapedPostRepository.findById(id)
.orElseThrow(() -> new EntityNotFoundException("게시물을 찾을 수 없습니다: " + id));
post.setViewedAt(LocalDateTime.now());
}
}
@@ -0,0 +1,44 @@
package company.specialsource.scraper;
import company.specialsource.entity.ScrapedPost;
import java.time.LocalDateTime;
import java.util.List;
public record ScrapedPostResponse(
Long id,
String boardKey,
String siteName,
String boardName,
String title,
String link,
String author,
String postedAt,
Integer viewCount,
Integer likeCount,
String content,
List<String> imageUrls,
LocalDateTime scrapedAt
) {
public static ScrapedPostResponse from(ScrapedPost post) {
List<String> imageUrls = (post.getImageUrls() == null || post.getImageUrls().isBlank())
? List.of()
: List.of(post.getImageUrls().split("\n"));
return new ScrapedPostResponse(
post.getId(),
post.getBoardKey(),
post.getSiteName(),
post.getBoardName(),
post.getTitle(),
post.getLink(),
post.getAuthor(),
post.getPostedAt(),
post.getViewCount(),
post.getLikeCount(),
post.getContent(),
imageUrls,
post.getScrapedAt()
);
}
}
@@ -0,0 +1,72 @@
package company.specialsource.scraper;
import company.specialsource.entity.ScrapedPost;
import company.specialsource.repository.ScrapedPostRepository;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.stereotype.Service;
import java.time.LocalDateTime;
import java.util.List;
/**
* 목록에 새로 나타난 글만 상세페이지까지 들어가서 본문을 수집한다.
* 이미 저장된 글은 상세 요청을 다시 하지 않으므로 매 주기마다 요청량이 쌓이지 않는다.
*/
@Slf4j
@Service
@RequiredArgsConstructor
public class ScrapedPostSyncService {
private static final long DELAY_BETWEEN_DETAIL_FETCHES_MS = 800;
private final ScrapedPostRepository scrapedPostRepository;
private final PostDetailScraperService postDetailScraperService;
public int saveNewPosts(BoardSource source, List<ScrapedPostDto> posts) {
int savedCount = 0;
for (ScrapedPostDto post : posts) {
if (scrapedPostRepository.existsByBoardKeyAndLink(source.name(), post.link())) {
continue;
}
PostDetail detail = fetchDetailSafely(source, post);
scrapedPostRepository.save(ScrapedPost.builder()
.boardKey(source.name())
.siteName(source.getSiteName())
.boardName(source.getBoardName())
.title(post.title())
.link(post.link())
.author(post.author())
.postedAt(post.postedAt())
.viewCount(post.viewCount())
.likeCount(post.likeCount())
.content(detail.content())
.imageUrls(String.join("\n", detail.imageUrls()))
.scrapedAt(LocalDateTime.now())
.build());
savedCount++;
sleepQuietly();
}
return savedCount;
}
private PostDetail fetchDetailSafely(BoardSource source, ScrapedPostDto post) {
try {
return postDetailScraperService.scrape(post.link(), source.getDetailParserType());
} catch (Exception e) {
log.warn("[본문 스크래핑 실패] {} - {}", post.link(), e.getMessage());
return PostDetail.EMPTY;
}
}
private void sleepQuietly() {
try {
Thread.sleep(DELAY_BETWEEN_DETAIL_FETCHES_MS);
} catch (InterruptedException e) {
Thread.currentThread().interrupt();
}
}
}
@@ -0,0 +1,73 @@
package company.specialsource.scraper;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Element;
import org.jsoup.safety.Safelist;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Set;
final class ScraperTextUtils {
private static final Safelist CONTENT_SAFELIST = Safelist.relaxed()
.addAttributes("img", "src", "alt", "referrerpolicy", "loading")
.addAttributes("a", "href", "target", "rel");
private ScraperTextUtils() {
}
static Integer parseIntSafe(String text) {
if (text == null) {
return null;
}
String digitsOnly = text.trim().replace(",", "");
if (!digitsOnly.matches("-?\\d+")) {
return null;
}
try {
return Integer.parseInt(digitsOnly);
} catch (NumberFormatException e) {
return null;
}
}
/**
* 본문 컨테이너에서 원본 게시물의 형태(문단, 굵게, 링크, 이미지 위치)를 보존한 채로
* 안전하게 걸러낸 HTML과 이미지 URL 목록을 뽑아낸다.
* img/a의 상대경로는 절대경로로 바꾼 뒤 Safelist로 스크립트 등 위험 요소만 제거한다.
*/
static PostDetail extractDetail(Element body) {
if (body == null) {
return PostDetail.EMPTY;
}
Element working = body.clone();
Set<String> imageUrls = new LinkedHashSet<>();
for (Element img : working.select("img")) {
String src = img.attr("abs:src");
if (src.isBlank()) {
img.remove();
continue;
}
img.attr("src", src);
img.attr("referrerpolicy", "no-referrer");
img.attr("loading", "lazy");
imageUrls.add(src);
}
for (Element link : working.select("a")) {
String href = link.attr("abs:href");
if (href.isBlank()) {
continue;
}
link.attr("href", href);
link.attr("target", "_blank");
link.attr("rel", "noopener noreferrer");
}
String contentHtml = Jsoup.clean(working.html(), working.baseUri(), CONTENT_SAFELIST).trim();
return new PostDetail(contentHtml, List.copyOf(imageUrls));
}
}
@@ -0,0 +1,46 @@
package company.specialsource.scraper;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.scheduling.annotation.Scheduled;
import org.springframework.stereotype.Component;
import java.util.List;
@Slf4j
@Component
@RequiredArgsConstructor
public class ScrapingScheduler {
// 사이트에 부담을 주지 않기 위해 게시판별 요청 사이 최소 간격을 둔다.
private static final long DELAY_BETWEEN_BOARDS_MS = 1500;
private final BoardScraperService boardScraperService;
private final ScrapedPostSyncService scrapedPostSyncService;
@Scheduled(
initialDelayString = "${scraping.initial-delay-ms:10000}",
fixedDelayString = "${scraping.interval-ms:600000}"
)
public void scrapeAllBoards() {
for (BoardSource source : BoardSource.values()) {
try {
List<ScrapedPostDto> posts = boardScraperService.scrape(source);
int savedCount = scrapedPostSyncService.saveNewPosts(source, posts);
log.info("[스크래핑] {}/{} - 수집 {}건, 신규 저장 {}건",
source.getSiteName(), source.getBoardName(), posts.size(), savedCount);
} catch (Exception e) {
log.warn("[스크래핑 실패] {}/{} - {}", source.getSiteName(), source.getBoardName(), e.getMessage());
}
sleepQuietly();
}
}
private void sleepQuietly() {
try {
Thread.sleep(DELAY_BETWEEN_BOARDS_MS);
} catch (InterruptedException e) {
Thread.currentThread().interrupt();
}
}
}