mirror of
https://github.com/headporter81/specialsource-homepage-backend.git
synced 2026-10-07 15:21:09 +09:00
chore : deploy 수정
This commit is contained in:
@@ -0,0 +1,9 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
public interface BoardParser {
|
||||
List<ScrapedPostDto> parse(Document document);
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import org.jsoup.Jsoup;
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.List;
|
||||
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class BoardScraperService {
|
||||
|
||||
private static final String USER_AGENT =
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36";
|
||||
private static final int TIMEOUT_MS = 10_000;
|
||||
|
||||
private final RuliwebBoardListParser ruliwebBoardListParser;
|
||||
private final RuliwebMypiParser ruliwebMypiParser;
|
||||
private final DcGalleryParser dcGalleryParser;
|
||||
|
||||
public List<ScrapedPostDto> scrape(BoardSource source) throws IOException {
|
||||
Document document = Jsoup.connect(source.getUrl())
|
||||
.userAgent(USER_AGENT)
|
||||
.timeout(TIMEOUT_MS)
|
||||
.get();
|
||||
|
||||
BoardParser parser = switch (source.getParserType()) {
|
||||
case RULIWEB_BOARD_LIST -> ruliwebBoardListParser;
|
||||
case RULIWEB_MYPI_BEST -> ruliwebMypiParser;
|
||||
case DC_GALLERY_LIST -> dcGalleryParser;
|
||||
};
|
||||
|
||||
return parser.parse(document);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import lombok.Getter;
|
||||
|
||||
@Getter
|
||||
public enum BoardSource {
|
||||
RULIWEB_BEST("루리웹", "실시간베스트", "https://bbs.ruliweb.com/best/selection", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
|
||||
RULIWEB_HUMOR("루리웹", "유머", "https://bbs.ruliweb.com/best/humor", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
|
||||
RULIWEB_MYPI("루리웹", "마이피", "https://mypi.ruliweb.com/", ParserType.RULIWEB_MYPI_BEST, DetailParserType.RULIWEB),
|
||||
RULIWEB_HOTDEAL("루리웹", "핫딜", "https://bbs.ruliweb.com/market/board/1020", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
|
||||
DC_BJJ("디씨인사이드", "주짓수 갤러리", "https://gall.dcinside.com/mgallery/board/lists/?id=bjj", ParserType.DC_GALLERY_LIST, DetailParserType.DC),
|
||||
DC_SINGULARITY("디씨인사이드", "특이점이 온다", "https://gall.dcinside.com/mgallery/board/lists?id=thesingularity", ParserType.DC_GALLERY_LIST, DetailParserType.DC);
|
||||
|
||||
private final String siteName;
|
||||
private final String boardName;
|
||||
private final String url;
|
||||
private final ParserType parserType;
|
||||
private final DetailParserType detailParserType;
|
||||
|
||||
BoardSource(String siteName, String boardName, String url, ParserType parserType, DetailParserType detailParserType) {
|
||||
this.siteName = siteName;
|
||||
this.boardName = boardName;
|
||||
this.url = url;
|
||||
this.parserType = parserType;
|
||||
this.detailParserType = detailParserType;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
public record BoardSummaryResponse(String boardKey, String siteName, String boardName) {
|
||||
public static BoardSummaryResponse from(BoardSource source) {
|
||||
return new BoardSummaryResponse(source.name(), source.getSiteName(), source.getBoardName());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.jsoup.nodes.Element;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* 디씨인사이드 갤러리 목록 파서. 공지/설문 행은 제외하고 일반 게시글만 수집한다.
|
||||
*/
|
||||
@Component
|
||||
public class DcGalleryParser implements BoardParser {
|
||||
|
||||
@Override
|
||||
public List<ScrapedPostDto> parse(Document document) {
|
||||
List<ScrapedPostDto> posts = new ArrayList<>();
|
||||
|
||||
for (Element row : document.select("tr.ub-content")) {
|
||||
if (!row.hasClass("us-post") || "icon_notice".equals(row.attr("data-type"))) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Element titleAnchor = row.selectFirst("td.gall_tit a");
|
||||
if (titleAnchor == null) {
|
||||
continue;
|
||||
}
|
||||
|
||||
String title = titleAnchor.text().trim();
|
||||
String link = titleAnchor.attr("abs:href");
|
||||
if (title.isEmpty() || link.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Element writerEl = row.selectFirst("td.gall_writer");
|
||||
String author = writerEl == null ? "" : writerEl.attr("data-nick");
|
||||
if (author.isBlank() && writerEl != null) {
|
||||
author = writerEl.text().trim();
|
||||
}
|
||||
|
||||
Element dateEl = row.selectFirst("td.gall_date");
|
||||
String postedAt = dateEl == null ? "" : dateEl.hasAttr("title") ? dateEl.attr("title") : dateEl.text().trim();
|
||||
|
||||
Integer viewCount = ScraperTextUtils.parseIntSafe(row.select("td.gall_count").text());
|
||||
Integer likeCount = ScraperTextUtils.parseIntSafe(row.select("td.gall_recommend").text());
|
||||
|
||||
posts.add(new ScrapedPostDto(title, link, author, postedAt, viewCount, likeCount));
|
||||
}
|
||||
|
||||
return posts;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
@Component
|
||||
public class DcPostDetailParser implements DetailParser {
|
||||
|
||||
@Override
|
||||
public PostDetail parse(Document document) {
|
||||
return ScraperTextUtils.extractDetail(document.selectFirst("div.write_div"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
|
||||
public interface DetailParser {
|
||||
PostDetail parse(Document document);
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
public enum DetailParserType {
|
||||
RULIWEB,
|
||||
DC
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
public enum ParserType {
|
||||
RULIWEB_BOARD_LIST, // 표 형태 게시판 (실시간베스트, 유머, 핫딜)
|
||||
RULIWEB_MYPI_BEST, // 마이피 실시간 인기글 위젯
|
||||
DC_GALLERY_LIST // 디씨인사이드 갤러리 게시판
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* content는 원본 게시물의 문단/굵게/링크/이미지 위치를 보존한 정제된(sanitized) HTML이다.
|
||||
*/
|
||||
public record PostDetail(String content, List<String> imageUrls) {
|
||||
static final PostDetail EMPTY = new PostDetail("", List.of());
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import org.jsoup.Jsoup;
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.io.IOException;
|
||||
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class PostDetailScraperService {
|
||||
|
||||
private static final String USER_AGENT =
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36";
|
||||
private static final int TIMEOUT_MS = 10_000;
|
||||
|
||||
private final RuliwebPostDetailParser ruliwebPostDetailParser;
|
||||
private final DcPostDetailParser dcPostDetailParser;
|
||||
|
||||
public PostDetail scrape(String url, DetailParserType type) throws IOException {
|
||||
Document document = Jsoup.connect(url)
|
||||
.userAgent(USER_AGENT)
|
||||
.timeout(TIMEOUT_MS)
|
||||
.get();
|
||||
|
||||
DetailParser parser = switch (type) {
|
||||
case RULIWEB -> ruliwebPostDetailParser;
|
||||
case DC -> dcPostDetailParser;
|
||||
};
|
||||
|
||||
return parser.parse(document);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.jsoup.nodes.Element;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* 루리웹 표 형태 게시판(실시간베스트, 유머, 핫딜) 공통 파서.
|
||||
* 게시판마다 tr.table_body 구조는 같고, 공지(class=notice) 행만 다르게 섞여 있어 걸러낸다.
|
||||
*/
|
||||
@Component
|
||||
public class RuliwebBoardListParser implements BoardParser {
|
||||
|
||||
@Override
|
||||
public List<ScrapedPostDto> parse(Document document) {
|
||||
List<ScrapedPostDto> posts = new ArrayList<>();
|
||||
|
||||
for (Element row : document.select("tr.table_body")) {
|
||||
if (row.hasClass("notice")) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Element titleAnchor = row.selectFirst("td.subject a");
|
||||
if (titleAnchor == null) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Element strong = titleAnchor.selectFirst("strong");
|
||||
String title = (strong != null ? strong.text() : titleAnchor.text()).trim();
|
||||
String link = titleAnchor.attr("abs:href");
|
||||
if (title.isEmpty() || link.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
String author = row.select("td.writer").text().trim();
|
||||
String postedAt = row.select("td.time").text().trim();
|
||||
Integer likeCount = ScraperTextUtils.parseIntSafe(row.select("td.recomd").text());
|
||||
Integer viewCount = ScraperTextUtils.parseIntSafe(row.select("td.hit").text());
|
||||
|
||||
posts.add(new ScrapedPostDto(title, link, author, postedAt, viewCount, likeCount));
|
||||
}
|
||||
|
||||
return posts;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.jsoup.nodes.Element;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* 마이피(mypi.ruliweb.com)의 "실시간 인기글" 위젯 파서.
|
||||
* 작성자/날짜/조회수는 목록에 노출되지 않아 title, link만 수집한다.
|
||||
*/
|
||||
@Component
|
||||
public class RuliwebMypiParser implements BoardParser {
|
||||
|
||||
@Override
|
||||
public List<ScrapedPostDto> parse(Document document) {
|
||||
List<ScrapedPostDto> posts = new ArrayList<>();
|
||||
|
||||
for (Element anchor : document.select("ul.right_best_list li.right_best_list_item a.txt_link")) {
|
||||
String title = anchor.text().trim();
|
||||
String link = anchor.attr("abs:href");
|
||||
if (title.isEmpty() || link.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
// 위젯 목록에 광고(외부 리다이렉트 링크)가 섞여 들어오는 경우가 있어 루리웹 도메인만 통과시킨다.
|
||||
if (!link.contains("ruliweb.com")) {
|
||||
continue;
|
||||
}
|
||||
posts.add(new ScrapedPostDto(title, link, null, null, null, null));
|
||||
}
|
||||
|
||||
return posts;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
@Component
|
||||
public class RuliwebPostDetailParser implements DetailParser {
|
||||
|
||||
@Override
|
||||
public PostDetail parse(Document document) {
|
||||
return ScraperTextUtils.extractDetail(document.selectFirst("div.view_content"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import company.specialsource.repository.ScrapedPostRepository;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.scheduling.annotation.Scheduled;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.time.LocalDateTime;
|
||||
|
||||
/**
|
||||
* 오래된 스크랩 게시물을 주기적으로 정리한다.
|
||||
* 스크래핑으로 계속 쌓이기만 하는 캐시성 데이터라 일정 기간이 지나면 삭제해도 무방하다.
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
public class ScrapedPostCleanupScheduler {
|
||||
|
||||
private final ScrapedPostRepository scrapedPostRepository;
|
||||
|
||||
@Value("${scraping.retention-days:7}")
|
||||
private int retentionDays;
|
||||
|
||||
@Scheduled(
|
||||
initialDelayString = "${scraping.cleanup-initial-delay-ms:60000}",
|
||||
fixedDelayString = "${scraping.cleanup-interval-ms:86400000}"
|
||||
)
|
||||
public void deleteExpiredPosts() {
|
||||
LocalDateTime cutoff = LocalDateTime.now().minusDays(retentionDays);
|
||||
long deletedCount = scrapedPostRepository.deleteByScrapedAtBefore(cutoff);
|
||||
if (deletedCount > 0) {
|
||||
log.info("[정리] 스크랩 {}일 경과 게시물 {}건 삭제 (기준: {})", retentionDays, deletedCount, cutoff);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
public record ScrapedPostDto(
|
||||
String title,
|
||||
String link,
|
||||
String author,
|
||||
String postedAt,
|
||||
Integer viewCount,
|
||||
Integer likeCount
|
||||
) {
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import company.specialsource.entity.ScrapedPost;
|
||||
import company.specialsource.repository.ScrapedPostRepository;
|
||||
import jakarta.persistence.EntityNotFoundException;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import org.springframework.data.domain.Page;
|
||||
import org.springframework.data.domain.Pageable;
|
||||
import org.springframework.stereotype.Service;
|
||||
import org.springframework.transaction.annotation.Transactional;
|
||||
|
||||
import java.time.LocalDateTime;
|
||||
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class ScrapedPostQueryService {
|
||||
|
||||
private final ScrapedPostRepository scrapedPostRepository;
|
||||
|
||||
public Page<ScrapedPostResponse> getPosts(String boardKey, Pageable pageable) {
|
||||
// 이미 조회한(viewedAt != null) 글은 목록에서 제외한다.
|
||||
Page<ScrapedPost> page = (boardKey == null || boardKey.isBlank())
|
||||
? scrapedPostRepository.findByViewedAtIsNullOrderByScrapedAtDesc(pageable)
|
||||
: scrapedPostRepository.findByBoardKeyAndViewedAtIsNullOrderByScrapedAtDesc(boardKey, pageable);
|
||||
|
||||
return page.map(ScrapedPostResponse::from);
|
||||
}
|
||||
|
||||
@Transactional
|
||||
public void markViewed(Long id) {
|
||||
ScrapedPost post = scrapedPostRepository.findById(id)
|
||||
.orElseThrow(() -> new EntityNotFoundException("게시물을 찾을 수 없습니다: " + id));
|
||||
post.setViewedAt(LocalDateTime.now());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import company.specialsource.entity.ScrapedPost;
|
||||
|
||||
import java.time.LocalDateTime;
|
||||
import java.util.List;
|
||||
|
||||
public record ScrapedPostResponse(
|
||||
Long id,
|
||||
String boardKey,
|
||||
String siteName,
|
||||
String boardName,
|
||||
String title,
|
||||
String link,
|
||||
String author,
|
||||
String postedAt,
|
||||
Integer viewCount,
|
||||
Integer likeCount,
|
||||
String content,
|
||||
List<String> imageUrls,
|
||||
LocalDateTime scrapedAt
|
||||
) {
|
||||
public static ScrapedPostResponse from(ScrapedPost post) {
|
||||
List<String> imageUrls = (post.getImageUrls() == null || post.getImageUrls().isBlank())
|
||||
? List.of()
|
||||
: List.of(post.getImageUrls().split("\n"));
|
||||
|
||||
return new ScrapedPostResponse(
|
||||
post.getId(),
|
||||
post.getBoardKey(),
|
||||
post.getSiteName(),
|
||||
post.getBoardName(),
|
||||
post.getTitle(),
|
||||
post.getLink(),
|
||||
post.getAuthor(),
|
||||
post.getPostedAt(),
|
||||
post.getViewCount(),
|
||||
post.getLikeCount(),
|
||||
post.getContent(),
|
||||
imageUrls,
|
||||
post.getScrapedAt()
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,72 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import company.specialsource.entity.ScrapedPost;
|
||||
import company.specialsource.repository.ScrapedPostRepository;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.time.LocalDateTime;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* 목록에 새로 나타난 글만 상세페이지까지 들어가서 본문을 수집한다.
|
||||
* 이미 저장된 글은 상세 요청을 다시 하지 않으므로 매 주기마다 요청량이 쌓이지 않는다.
|
||||
*/
|
||||
@Slf4j
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class ScrapedPostSyncService {
|
||||
|
||||
private static final long DELAY_BETWEEN_DETAIL_FETCHES_MS = 800;
|
||||
|
||||
private final ScrapedPostRepository scrapedPostRepository;
|
||||
private final PostDetailScraperService postDetailScraperService;
|
||||
|
||||
public int saveNewPosts(BoardSource source, List<ScrapedPostDto> posts) {
|
||||
int savedCount = 0;
|
||||
for (ScrapedPostDto post : posts) {
|
||||
if (scrapedPostRepository.existsByBoardKeyAndLink(source.name(), post.link())) {
|
||||
continue;
|
||||
}
|
||||
|
||||
PostDetail detail = fetchDetailSafely(source, post);
|
||||
|
||||
scrapedPostRepository.save(ScrapedPost.builder()
|
||||
.boardKey(source.name())
|
||||
.siteName(source.getSiteName())
|
||||
.boardName(source.getBoardName())
|
||||
.title(post.title())
|
||||
.link(post.link())
|
||||
.author(post.author())
|
||||
.postedAt(post.postedAt())
|
||||
.viewCount(post.viewCount())
|
||||
.likeCount(post.likeCount())
|
||||
.content(detail.content())
|
||||
.imageUrls(String.join("\n", detail.imageUrls()))
|
||||
.scrapedAt(LocalDateTime.now())
|
||||
.build());
|
||||
savedCount++;
|
||||
|
||||
sleepQuietly();
|
||||
}
|
||||
return savedCount;
|
||||
}
|
||||
|
||||
private PostDetail fetchDetailSafely(BoardSource source, ScrapedPostDto post) {
|
||||
try {
|
||||
return postDetailScraperService.scrape(post.link(), source.getDetailParserType());
|
||||
} catch (Exception e) {
|
||||
log.warn("[본문 스크래핑 실패] {} - {}", post.link(), e.getMessage());
|
||||
return PostDetail.EMPTY;
|
||||
}
|
||||
}
|
||||
|
||||
private void sleepQuietly() {
|
||||
try {
|
||||
Thread.sleep(DELAY_BETWEEN_DETAIL_FETCHES_MS);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.Jsoup;
|
||||
import org.jsoup.nodes.Element;
|
||||
import org.jsoup.safety.Safelist;
|
||||
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
final class ScraperTextUtils {
|
||||
|
||||
private static final Safelist CONTENT_SAFELIST = Safelist.relaxed()
|
||||
.addAttributes("img", "src", "alt", "referrerpolicy", "loading")
|
||||
.addAttributes("a", "href", "target", "rel");
|
||||
|
||||
private ScraperTextUtils() {
|
||||
}
|
||||
|
||||
static Integer parseIntSafe(String text) {
|
||||
if (text == null) {
|
||||
return null;
|
||||
}
|
||||
String digitsOnly = text.trim().replace(",", "");
|
||||
if (!digitsOnly.matches("-?\\d+")) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
return Integer.parseInt(digitsOnly);
|
||||
} catch (NumberFormatException e) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 본문 컨테이너에서 원본 게시물의 형태(문단, 굵게, 링크, 이미지 위치)를 보존한 채로
|
||||
* 안전하게 걸러낸 HTML과 이미지 URL 목록을 뽑아낸다.
|
||||
* img/a의 상대경로는 절대경로로 바꾼 뒤 Safelist로 스크립트 등 위험 요소만 제거한다.
|
||||
*/
|
||||
static PostDetail extractDetail(Element body) {
|
||||
if (body == null) {
|
||||
return PostDetail.EMPTY;
|
||||
}
|
||||
|
||||
Element working = body.clone();
|
||||
|
||||
Set<String> imageUrls = new LinkedHashSet<>();
|
||||
for (Element img : working.select("img")) {
|
||||
String src = img.attr("abs:src");
|
||||
if (src.isBlank()) {
|
||||
img.remove();
|
||||
continue;
|
||||
}
|
||||
img.attr("src", src);
|
||||
img.attr("referrerpolicy", "no-referrer");
|
||||
img.attr("loading", "lazy");
|
||||
imageUrls.add(src);
|
||||
}
|
||||
for (Element link : working.select("a")) {
|
||||
String href = link.attr("abs:href");
|
||||
if (href.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
link.attr("href", href);
|
||||
link.attr("target", "_blank");
|
||||
link.attr("rel", "noopener noreferrer");
|
||||
}
|
||||
|
||||
String contentHtml = Jsoup.clean(working.html(), working.baseUri(), CONTENT_SAFELIST).trim();
|
||||
|
||||
return new PostDetail(contentHtml, List.copyOf(imageUrls));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.scheduling.annotation.Scheduled;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
@Slf4j
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
public class ScrapingScheduler {
|
||||
|
||||
// 사이트에 부담을 주지 않기 위해 게시판별 요청 사이 최소 간격을 둔다.
|
||||
private static final long DELAY_BETWEEN_BOARDS_MS = 1500;
|
||||
|
||||
private final BoardScraperService boardScraperService;
|
||||
private final ScrapedPostSyncService scrapedPostSyncService;
|
||||
|
||||
@Scheduled(
|
||||
initialDelayString = "${scraping.initial-delay-ms:10000}",
|
||||
fixedDelayString = "${scraping.interval-ms:600000}"
|
||||
)
|
||||
public void scrapeAllBoards() {
|
||||
for (BoardSource source : BoardSource.values()) {
|
||||
try {
|
||||
List<ScrapedPostDto> posts = boardScraperService.scrape(source);
|
||||
int savedCount = scrapedPostSyncService.saveNewPosts(source, posts);
|
||||
log.info("[스크래핑] {}/{} - 수집 {}건, 신규 저장 {}건",
|
||||
source.getSiteName(), source.getBoardName(), posts.size(), savedCount);
|
||||
} catch (Exception e) {
|
||||
log.warn("[스크래핑 실패] {}/{} - {}", source.getSiteName(), source.getBoardName(), e.getMessage());
|
||||
}
|
||||
sleepQuietly();
|
||||
}
|
||||
}
|
||||
|
||||
private void sleepQuietly() {
|
||||
try {
|
||||
Thread.sleep(DELAY_BETWEEN_BOARDS_MS);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user