mirror of
https://github.com/headporter81/specialsource-homepage-backend.git
synced 2026-10-07 15:21:09 +09:00
chore : deploy 수정
This commit is contained in:
@@ -2,8 +2,10 @@ package company.specialsource;
|
||||
|
||||
import org.springframework.boot.SpringApplication;
|
||||
import org.springframework.boot.autoconfigure.SpringBootApplication;
|
||||
import org.springframework.scheduling.annotation.EnableScheduling;
|
||||
|
||||
@SpringBootApplication
|
||||
@EnableScheduling
|
||||
public class SpecialsourceHomepageApplication {
|
||||
|
||||
public static void main(String[] args) {
|
||||
|
||||
@@ -21,7 +21,7 @@ public class DataInitializer implements CommandLineRunner {
|
||||
if (userRepository.findByUsername("admin").isEmpty()) {
|
||||
userRepository.save(User.builder()
|
||||
.username("admin")
|
||||
.password(passwordEncoder.encode("1234"))
|
||||
.password(passwordEncoder.encode("Ppasse444$"))
|
||||
.role("ROLE_USER")
|
||||
.build());
|
||||
log.info("초기 관리자 계정(admin)이 생성되었습니다.");
|
||||
|
||||
@@ -0,0 +1,46 @@
|
||||
package company.specialsource.config;
|
||||
|
||||
import jakarta.servlet.FilterChain;
|
||||
import jakarta.servlet.ServletException;
|
||||
import jakarta.servlet.http.HttpServletRequest;
|
||||
import jakarta.servlet.http.HttpServletResponse;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import org.springframework.security.authentication.UsernamePasswordAuthenticationToken;
|
||||
import org.springframework.security.core.context.SecurityContextHolder;
|
||||
import org.springframework.stereotype.Component;
|
||||
import org.springframework.web.filter.OncePerRequestFilter;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.List;
|
||||
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
public class JwtAuthenticationFilter extends OncePerRequestFilter {
|
||||
|
||||
private static final String HEADER_PREFIX = "Bearer ";
|
||||
|
||||
private final JwtTokenProvider jwtTokenProvider;
|
||||
|
||||
@Override
|
||||
protected void doFilterInternal(
|
||||
HttpServletRequest request,
|
||||
HttpServletResponse response,
|
||||
FilterChain filterChain
|
||||
) throws ServletException, IOException {
|
||||
|
||||
String header = request.getHeader("Authorization");
|
||||
|
||||
if (header != null && header.startsWith(HEADER_PREFIX)) {
|
||||
String token = header.substring(HEADER_PREFIX.length());
|
||||
|
||||
if (jwtTokenProvider.validateToken(token)) {
|
||||
String username = jwtTokenProvider.getUsername(token);
|
||||
UsernamePasswordAuthenticationToken authentication =
|
||||
new UsernamePasswordAuthenticationToken(username, null, List.of());
|
||||
SecurityContextHolder.getContext().setAuthentication(authentication);
|
||||
}
|
||||
}
|
||||
|
||||
filterChain.doFilter(request, response);
|
||||
}
|
||||
}
|
||||
@@ -1,25 +1,33 @@
|
||||
package company.specialsource.config;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import org.springframework.context.annotation.Bean;
|
||||
import org.springframework.context.annotation.Configuration;
|
||||
import org.springframework.security.config.annotation.web.builders.HttpSecurity;
|
||||
import org.springframework.security.config.annotation.web.configuration.EnableWebSecurity;
|
||||
import org.springframework.security.config.http.SessionCreationPolicy;
|
||||
import org.springframework.security.crypto.bcrypt.BCryptPasswordEncoder;
|
||||
import org.springframework.security.crypto.password.PasswordEncoder;
|
||||
import org.springframework.security.web.SecurityFilterChain;
|
||||
import org.springframework.security.web.authentication.UsernamePasswordAuthenticationFilter;
|
||||
|
||||
@Configuration
|
||||
@EnableWebSecurity
|
||||
@RequiredArgsConstructor
|
||||
public class SecurityConfig {
|
||||
|
||||
private final JwtAuthenticationFilter jwtAuthenticationFilter;
|
||||
|
||||
@Bean
|
||||
public SecurityFilterChain filterChain(HttpSecurity http) throws Exception {
|
||||
http
|
||||
.csrf(csrf -> csrf.disable()) // REST API 통신을 위해 CSRF 비활성화
|
||||
.sessionManagement(session -> session.sessionCreationPolicy(SessionCreationPolicy.STATELESS))
|
||||
.authorizeHttpRequests(auth -> auth
|
||||
.requestMatchers("/api/login").permitAll() // 로그인 API는 누구나 접근 가능
|
||||
.anyRequest().authenticated()
|
||||
);
|
||||
)
|
||||
.addFilterBefore(jwtAuthenticationFilter, UsernamePasswordAuthenticationFilter.class);
|
||||
return http.build();
|
||||
}
|
||||
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
package company.specialsource.controller;
|
||||
|
||||
import company.specialsource.scraper.BoardSource;
|
||||
import company.specialsource.scraper.BoardSummaryResponse;
|
||||
import company.specialsource.scraper.ScrapedPostQueryService;
|
||||
import company.specialsource.scraper.ScrapedPostResponse;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import org.springframework.data.domain.Page;
|
||||
import org.springframework.data.domain.PageRequest;
|
||||
import org.springframework.data.domain.Pageable;
|
||||
import org.springframework.http.ResponseEntity;
|
||||
import org.springframework.web.bind.annotation.*;
|
||||
|
||||
import java.util.Arrays;
|
||||
import java.util.List;
|
||||
|
||||
@RestController
|
||||
@RequestMapping("/api/scraped-posts")
|
||||
@RequiredArgsConstructor
|
||||
public class ScrapedPostController {
|
||||
|
||||
private final ScrapedPostQueryService scrapedPostQueryService;
|
||||
|
||||
@GetMapping
|
||||
public ResponseEntity<?> getPosts(
|
||||
@RequestParam(required = false) String board,
|
||||
@RequestParam(defaultValue = "0") int page,
|
||||
@RequestParam(defaultValue = "20") int size
|
||||
) {
|
||||
if (board != null && !isValidBoardKey(board)) {
|
||||
return ResponseEntity.badRequest().body("존재하지 않는 게시판 키입니다: " + board);
|
||||
}
|
||||
|
||||
Pageable pageable = PageRequest.of(page, Math.min(size, 100));
|
||||
Page<ScrapedPostResponse> posts = scrapedPostQueryService.getPosts(board, pageable);
|
||||
return ResponseEntity.ok(posts);
|
||||
}
|
||||
|
||||
@GetMapping("/boards")
|
||||
public ResponseEntity<List<BoardSummaryResponse>> getBoards() {
|
||||
List<BoardSummaryResponse> boards = Arrays.stream(BoardSource.values())
|
||||
.map(BoardSummaryResponse::from)
|
||||
.toList();
|
||||
return ResponseEntity.ok(boards);
|
||||
}
|
||||
|
||||
@PostMapping("/{id}/view")
|
||||
public ResponseEntity<Void> markViewed(@PathVariable Long id) {
|
||||
scrapedPostQueryService.markViewed(id);
|
||||
return ResponseEntity.noContent().build();
|
||||
}
|
||||
|
||||
private boolean isValidBoardKey(String boardKey) {
|
||||
return Arrays.stream(BoardSource.values()).anyMatch(source -> source.name().equals(boardKey));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,56 @@
|
||||
package company.specialsource.entity;
|
||||
|
||||
import jakarta.persistence.*;
|
||||
import lombok.*;
|
||||
|
||||
import java.time.LocalDateTime;
|
||||
|
||||
@Entity
|
||||
@Table(name = "scraped_posts",
|
||||
uniqueConstraints = @UniqueConstraint(columnNames = {"board_key", "link"}))
|
||||
@Getter
|
||||
@NoArgsConstructor(access = AccessLevel.PROTECTED)
|
||||
@AllArgsConstructor
|
||||
@Builder
|
||||
public class ScrapedPost {
|
||||
|
||||
@Id
|
||||
@GeneratedValue(strategy = GenerationType.IDENTITY)
|
||||
private Long id;
|
||||
|
||||
@Column(name = "board_key", nullable = false)
|
||||
private String boardKey; // BoardSource enum name, 예: RULIWEB_BEST
|
||||
|
||||
@Column(name = "site_name", nullable = false)
|
||||
private String siteName; // 예: 루리웹, 디씨인사이드
|
||||
|
||||
@Column(name = "board_name", nullable = false)
|
||||
private String boardName; // 예: 실시간베스트, 주짓수 갤러리
|
||||
|
||||
@Column(nullable = false, length = 500)
|
||||
private String title;
|
||||
|
||||
@Column(nullable = false, length = 1000)
|
||||
private String link;
|
||||
|
||||
private String author;
|
||||
|
||||
private String postedAt; // 사이트마다 표기 형식이 달라(HH:mm, yy.MM.dd 등) 원문 그대로 저장
|
||||
|
||||
private Integer viewCount;
|
||||
|
||||
private Integer likeCount;
|
||||
|
||||
@Column(columnDefinition = "TEXT")
|
||||
private String content; // 게시물 상세페이지에서 추출한 본문 (정제된 HTML, 원본 문단/링크/이미지 위치 보존)
|
||||
|
||||
@Column(name = "image_urls", columnDefinition = "TEXT")
|
||||
private String imageUrls; // 본문에 포함된 이미지 URL, 줄바꿈으로 구분
|
||||
|
||||
@Column(name = "scraped_at", nullable = false)
|
||||
private LocalDateTime scrapedAt;
|
||||
|
||||
@Setter
|
||||
@Column(name = "viewed_at")
|
||||
private LocalDateTime viewedAt; // 게시판 조회 시각. null이면 아직 안 읽은 글
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
package company.specialsource.repository;
|
||||
|
||||
import company.specialsource.entity.ScrapedPost;
|
||||
import org.springframework.data.domain.Page;
|
||||
import org.springframework.data.domain.Pageable;
|
||||
import org.springframework.data.jpa.repository.JpaRepository;
|
||||
import org.springframework.transaction.annotation.Transactional;
|
||||
|
||||
import java.time.LocalDateTime;
|
||||
|
||||
public interface ScrapedPostRepository extends JpaRepository<ScrapedPost, Long> {
|
||||
boolean existsByBoardKeyAndLink(String boardKey, String link);
|
||||
|
||||
Page<ScrapedPost> findByBoardKeyAndViewedAtIsNullOrderByScrapedAtDesc(String boardKey, Pageable pageable);
|
||||
|
||||
Page<ScrapedPost> findByViewedAtIsNullOrderByScrapedAtDesc(Pageable pageable);
|
||||
|
||||
@Transactional
|
||||
long deleteByScrapedAtBefore(LocalDateTime cutoff);
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
public interface BoardParser {
|
||||
List<ScrapedPostDto> parse(Document document);
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import org.jsoup.Jsoup;
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.List;
|
||||
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class BoardScraperService {
|
||||
|
||||
private static final String USER_AGENT =
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36";
|
||||
private static final int TIMEOUT_MS = 10_000;
|
||||
|
||||
private final RuliwebBoardListParser ruliwebBoardListParser;
|
||||
private final RuliwebMypiParser ruliwebMypiParser;
|
||||
private final DcGalleryParser dcGalleryParser;
|
||||
|
||||
public List<ScrapedPostDto> scrape(BoardSource source) throws IOException {
|
||||
Document document = Jsoup.connect(source.getUrl())
|
||||
.userAgent(USER_AGENT)
|
||||
.timeout(TIMEOUT_MS)
|
||||
.get();
|
||||
|
||||
BoardParser parser = switch (source.getParserType()) {
|
||||
case RULIWEB_BOARD_LIST -> ruliwebBoardListParser;
|
||||
case RULIWEB_MYPI_BEST -> ruliwebMypiParser;
|
||||
case DC_GALLERY_LIST -> dcGalleryParser;
|
||||
};
|
||||
|
||||
return parser.parse(document);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import lombok.Getter;
|
||||
|
||||
@Getter
|
||||
public enum BoardSource {
|
||||
RULIWEB_BEST("루리웹", "실시간베스트", "https://bbs.ruliweb.com/best/selection", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
|
||||
RULIWEB_HUMOR("루리웹", "유머", "https://bbs.ruliweb.com/best/humor", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
|
||||
RULIWEB_MYPI("루리웹", "마이피", "https://mypi.ruliweb.com/", ParserType.RULIWEB_MYPI_BEST, DetailParserType.RULIWEB),
|
||||
RULIWEB_HOTDEAL("루리웹", "핫딜", "https://bbs.ruliweb.com/market/board/1020", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
|
||||
DC_BJJ("디씨인사이드", "주짓수 갤러리", "https://gall.dcinside.com/mgallery/board/lists/?id=bjj", ParserType.DC_GALLERY_LIST, DetailParserType.DC),
|
||||
DC_SINGULARITY("디씨인사이드", "특이점이 온다", "https://gall.dcinside.com/mgallery/board/lists?id=thesingularity", ParserType.DC_GALLERY_LIST, DetailParserType.DC);
|
||||
|
||||
private final String siteName;
|
||||
private final String boardName;
|
||||
private final String url;
|
||||
private final ParserType parserType;
|
||||
private final DetailParserType detailParserType;
|
||||
|
||||
BoardSource(String siteName, String boardName, String url, ParserType parserType, DetailParserType detailParserType) {
|
||||
this.siteName = siteName;
|
||||
this.boardName = boardName;
|
||||
this.url = url;
|
||||
this.parserType = parserType;
|
||||
this.detailParserType = detailParserType;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
public record BoardSummaryResponse(String boardKey, String siteName, String boardName) {
|
||||
public static BoardSummaryResponse from(BoardSource source) {
|
||||
return new BoardSummaryResponse(source.name(), source.getSiteName(), source.getBoardName());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.jsoup.nodes.Element;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* 디씨인사이드 갤러리 목록 파서. 공지/설문 행은 제외하고 일반 게시글만 수집한다.
|
||||
*/
|
||||
@Component
|
||||
public class DcGalleryParser implements BoardParser {
|
||||
|
||||
@Override
|
||||
public List<ScrapedPostDto> parse(Document document) {
|
||||
List<ScrapedPostDto> posts = new ArrayList<>();
|
||||
|
||||
for (Element row : document.select("tr.ub-content")) {
|
||||
if (!row.hasClass("us-post") || "icon_notice".equals(row.attr("data-type"))) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Element titleAnchor = row.selectFirst("td.gall_tit a");
|
||||
if (titleAnchor == null) {
|
||||
continue;
|
||||
}
|
||||
|
||||
String title = titleAnchor.text().trim();
|
||||
String link = titleAnchor.attr("abs:href");
|
||||
if (title.isEmpty() || link.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Element writerEl = row.selectFirst("td.gall_writer");
|
||||
String author = writerEl == null ? "" : writerEl.attr("data-nick");
|
||||
if (author.isBlank() && writerEl != null) {
|
||||
author = writerEl.text().trim();
|
||||
}
|
||||
|
||||
Element dateEl = row.selectFirst("td.gall_date");
|
||||
String postedAt = dateEl == null ? "" : dateEl.hasAttr("title") ? dateEl.attr("title") : dateEl.text().trim();
|
||||
|
||||
Integer viewCount = ScraperTextUtils.parseIntSafe(row.select("td.gall_count").text());
|
||||
Integer likeCount = ScraperTextUtils.parseIntSafe(row.select("td.gall_recommend").text());
|
||||
|
||||
posts.add(new ScrapedPostDto(title, link, author, postedAt, viewCount, likeCount));
|
||||
}
|
||||
|
||||
return posts;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
@Component
|
||||
public class DcPostDetailParser implements DetailParser {
|
||||
|
||||
@Override
|
||||
public PostDetail parse(Document document) {
|
||||
return ScraperTextUtils.extractDetail(document.selectFirst("div.write_div"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
|
||||
public interface DetailParser {
|
||||
PostDetail parse(Document document);
|
||||
}
|
||||
@@ -0,0 +1,6 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
public enum DetailParserType {
|
||||
RULIWEB,
|
||||
DC
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
public enum ParserType {
|
||||
RULIWEB_BOARD_LIST, // 표 형태 게시판 (실시간베스트, 유머, 핫딜)
|
||||
RULIWEB_MYPI_BEST, // 마이피 실시간 인기글 위젯
|
||||
DC_GALLERY_LIST // 디씨인사이드 갤러리 게시판
|
||||
}
|
||||
@@ -0,0 +1,10 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* content는 원본 게시물의 문단/굵게/링크/이미지 위치를 보존한 정제된(sanitized) HTML이다.
|
||||
*/
|
||||
public record PostDetail(String content, List<String> imageUrls) {
|
||||
static final PostDetail EMPTY = new PostDetail("", List.of());
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import org.jsoup.Jsoup;
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.io.IOException;
|
||||
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class PostDetailScraperService {
|
||||
|
||||
private static final String USER_AGENT =
|
||||
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36";
|
||||
private static final int TIMEOUT_MS = 10_000;
|
||||
|
||||
private final RuliwebPostDetailParser ruliwebPostDetailParser;
|
||||
private final DcPostDetailParser dcPostDetailParser;
|
||||
|
||||
public PostDetail scrape(String url, DetailParserType type) throws IOException {
|
||||
Document document = Jsoup.connect(url)
|
||||
.userAgent(USER_AGENT)
|
||||
.timeout(TIMEOUT_MS)
|
||||
.get();
|
||||
|
||||
DetailParser parser = switch (type) {
|
||||
case RULIWEB -> ruliwebPostDetailParser;
|
||||
case DC -> dcPostDetailParser;
|
||||
};
|
||||
|
||||
return parser.parse(document);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,48 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.jsoup.nodes.Element;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* 루리웹 표 형태 게시판(실시간베스트, 유머, 핫딜) 공통 파서.
|
||||
* 게시판마다 tr.table_body 구조는 같고, 공지(class=notice) 행만 다르게 섞여 있어 걸러낸다.
|
||||
*/
|
||||
@Component
|
||||
public class RuliwebBoardListParser implements BoardParser {
|
||||
|
||||
@Override
|
||||
public List<ScrapedPostDto> parse(Document document) {
|
||||
List<ScrapedPostDto> posts = new ArrayList<>();
|
||||
|
||||
for (Element row : document.select("tr.table_body")) {
|
||||
if (row.hasClass("notice")) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Element titleAnchor = row.selectFirst("td.subject a");
|
||||
if (titleAnchor == null) {
|
||||
continue;
|
||||
}
|
||||
|
||||
Element strong = titleAnchor.selectFirst("strong");
|
||||
String title = (strong != null ? strong.text() : titleAnchor.text()).trim();
|
||||
String link = titleAnchor.attr("abs:href");
|
||||
if (title.isEmpty() || link.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
|
||||
String author = row.select("td.writer").text().trim();
|
||||
String postedAt = row.select("td.time").text().trim();
|
||||
Integer likeCount = ScraperTextUtils.parseIntSafe(row.select("td.recomd").text());
|
||||
Integer viewCount = ScraperTextUtils.parseIntSafe(row.select("td.hit").text());
|
||||
|
||||
posts.add(new ScrapedPostDto(title, link, author, postedAt, viewCount, likeCount));
|
||||
}
|
||||
|
||||
return posts;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.jsoup.nodes.Element;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* 마이피(mypi.ruliweb.com)의 "실시간 인기글" 위젯 파서.
|
||||
* 작성자/날짜/조회수는 목록에 노출되지 않아 title, link만 수집한다.
|
||||
*/
|
||||
@Component
|
||||
public class RuliwebMypiParser implements BoardParser {
|
||||
|
||||
@Override
|
||||
public List<ScrapedPostDto> parse(Document document) {
|
||||
List<ScrapedPostDto> posts = new ArrayList<>();
|
||||
|
||||
for (Element anchor : document.select("ul.right_best_list li.right_best_list_item a.txt_link")) {
|
||||
String title = anchor.text().trim();
|
||||
String link = anchor.attr("abs:href");
|
||||
if (title.isEmpty() || link.isEmpty()) {
|
||||
continue;
|
||||
}
|
||||
// 위젯 목록에 광고(외부 리다이렉트 링크)가 섞여 들어오는 경우가 있어 루리웹 도메인만 통과시킨다.
|
||||
if (!link.contains("ruliweb.com")) {
|
||||
continue;
|
||||
}
|
||||
posts.add(new ScrapedPostDto(title, link, null, null, null, null));
|
||||
}
|
||||
|
||||
return posts;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
@Component
|
||||
public class RuliwebPostDetailParser implements DetailParser {
|
||||
|
||||
@Override
|
||||
public PostDetail parse(Document document) {
|
||||
return ScraperTextUtils.extractDetail(document.selectFirst("div.view_content"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import company.specialsource.repository.ScrapedPostRepository;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.scheduling.annotation.Scheduled;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.time.LocalDateTime;
|
||||
|
||||
/**
|
||||
* 오래된 스크랩 게시물을 주기적으로 정리한다.
|
||||
* 스크래핑으로 계속 쌓이기만 하는 캐시성 데이터라 일정 기간이 지나면 삭제해도 무방하다.
|
||||
*/
|
||||
@Slf4j
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
public class ScrapedPostCleanupScheduler {
|
||||
|
||||
private final ScrapedPostRepository scrapedPostRepository;
|
||||
|
||||
@Value("${scraping.retention-days:7}")
|
||||
private int retentionDays;
|
||||
|
||||
@Scheduled(
|
||||
initialDelayString = "${scraping.cleanup-initial-delay-ms:60000}",
|
||||
fixedDelayString = "${scraping.cleanup-interval-ms:86400000}"
|
||||
)
|
||||
public void deleteExpiredPosts() {
|
||||
LocalDateTime cutoff = LocalDateTime.now().minusDays(retentionDays);
|
||||
long deletedCount = scrapedPostRepository.deleteByScrapedAtBefore(cutoff);
|
||||
if (deletedCount > 0) {
|
||||
log.info("[정리] 스크랩 {}일 경과 게시물 {}건 삭제 (기준: {})", retentionDays, deletedCount, cutoff);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
public record ScrapedPostDto(
|
||||
String title,
|
||||
String link,
|
||||
String author,
|
||||
String postedAt,
|
||||
Integer viewCount,
|
||||
Integer likeCount
|
||||
) {
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import company.specialsource.entity.ScrapedPost;
|
||||
import company.specialsource.repository.ScrapedPostRepository;
|
||||
import jakarta.persistence.EntityNotFoundException;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import org.springframework.data.domain.Page;
|
||||
import org.springframework.data.domain.Pageable;
|
||||
import org.springframework.stereotype.Service;
|
||||
import org.springframework.transaction.annotation.Transactional;
|
||||
|
||||
import java.time.LocalDateTime;
|
||||
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class ScrapedPostQueryService {
|
||||
|
||||
private final ScrapedPostRepository scrapedPostRepository;
|
||||
|
||||
public Page<ScrapedPostResponse> getPosts(String boardKey, Pageable pageable) {
|
||||
// 이미 조회한(viewedAt != null) 글은 목록에서 제외한다.
|
||||
Page<ScrapedPost> page = (boardKey == null || boardKey.isBlank())
|
||||
? scrapedPostRepository.findByViewedAtIsNullOrderByScrapedAtDesc(pageable)
|
||||
: scrapedPostRepository.findByBoardKeyAndViewedAtIsNullOrderByScrapedAtDesc(boardKey, pageable);
|
||||
|
||||
return page.map(ScrapedPostResponse::from);
|
||||
}
|
||||
|
||||
@Transactional
|
||||
public void markViewed(Long id) {
|
||||
ScrapedPost post = scrapedPostRepository.findById(id)
|
||||
.orElseThrow(() -> new EntityNotFoundException("게시물을 찾을 수 없습니다: " + id));
|
||||
post.setViewedAt(LocalDateTime.now());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import company.specialsource.entity.ScrapedPost;
|
||||
|
||||
import java.time.LocalDateTime;
|
||||
import java.util.List;
|
||||
|
||||
public record ScrapedPostResponse(
|
||||
Long id,
|
||||
String boardKey,
|
||||
String siteName,
|
||||
String boardName,
|
||||
String title,
|
||||
String link,
|
||||
String author,
|
||||
String postedAt,
|
||||
Integer viewCount,
|
||||
Integer likeCount,
|
||||
String content,
|
||||
List<String> imageUrls,
|
||||
LocalDateTime scrapedAt
|
||||
) {
|
||||
public static ScrapedPostResponse from(ScrapedPost post) {
|
||||
List<String> imageUrls = (post.getImageUrls() == null || post.getImageUrls().isBlank())
|
||||
? List.of()
|
||||
: List.of(post.getImageUrls().split("\n"));
|
||||
|
||||
return new ScrapedPostResponse(
|
||||
post.getId(),
|
||||
post.getBoardKey(),
|
||||
post.getSiteName(),
|
||||
post.getBoardName(),
|
||||
post.getTitle(),
|
||||
post.getLink(),
|
||||
post.getAuthor(),
|
||||
post.getPostedAt(),
|
||||
post.getViewCount(),
|
||||
post.getLikeCount(),
|
||||
post.getContent(),
|
||||
imageUrls,
|
||||
post.getScrapedAt()
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,72 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import company.specialsource.entity.ScrapedPost;
|
||||
import company.specialsource.repository.ScrapedPostRepository;
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.stereotype.Service;
|
||||
|
||||
import java.time.LocalDateTime;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* 목록에 새로 나타난 글만 상세페이지까지 들어가서 본문을 수집한다.
|
||||
* 이미 저장된 글은 상세 요청을 다시 하지 않으므로 매 주기마다 요청량이 쌓이지 않는다.
|
||||
*/
|
||||
@Slf4j
|
||||
@Service
|
||||
@RequiredArgsConstructor
|
||||
public class ScrapedPostSyncService {
|
||||
|
||||
private static final long DELAY_BETWEEN_DETAIL_FETCHES_MS = 800;
|
||||
|
||||
private final ScrapedPostRepository scrapedPostRepository;
|
||||
private final PostDetailScraperService postDetailScraperService;
|
||||
|
||||
public int saveNewPosts(BoardSource source, List<ScrapedPostDto> posts) {
|
||||
int savedCount = 0;
|
||||
for (ScrapedPostDto post : posts) {
|
||||
if (scrapedPostRepository.existsByBoardKeyAndLink(source.name(), post.link())) {
|
||||
continue;
|
||||
}
|
||||
|
||||
PostDetail detail = fetchDetailSafely(source, post);
|
||||
|
||||
scrapedPostRepository.save(ScrapedPost.builder()
|
||||
.boardKey(source.name())
|
||||
.siteName(source.getSiteName())
|
||||
.boardName(source.getBoardName())
|
||||
.title(post.title())
|
||||
.link(post.link())
|
||||
.author(post.author())
|
||||
.postedAt(post.postedAt())
|
||||
.viewCount(post.viewCount())
|
||||
.likeCount(post.likeCount())
|
||||
.content(detail.content())
|
||||
.imageUrls(String.join("\n", detail.imageUrls()))
|
||||
.scrapedAt(LocalDateTime.now())
|
||||
.build());
|
||||
savedCount++;
|
||||
|
||||
sleepQuietly();
|
||||
}
|
||||
return savedCount;
|
||||
}
|
||||
|
||||
private PostDetail fetchDetailSafely(BoardSource source, ScrapedPostDto post) {
|
||||
try {
|
||||
return postDetailScraperService.scrape(post.link(), source.getDetailParserType());
|
||||
} catch (Exception e) {
|
||||
log.warn("[본문 스크래핑 실패] {} - {}", post.link(), e.getMessage());
|
||||
return PostDetail.EMPTY;
|
||||
}
|
||||
}
|
||||
|
||||
private void sleepQuietly() {
|
||||
try {
|
||||
Thread.sleep(DELAY_BETWEEN_DETAIL_FETCHES_MS);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.Jsoup;
|
||||
import org.jsoup.nodes.Element;
|
||||
import org.jsoup.safety.Safelist;
|
||||
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
final class ScraperTextUtils {
|
||||
|
||||
private static final Safelist CONTENT_SAFELIST = Safelist.relaxed()
|
||||
.addAttributes("img", "src", "alt", "referrerpolicy", "loading")
|
||||
.addAttributes("a", "href", "target", "rel");
|
||||
|
||||
private ScraperTextUtils() {
|
||||
}
|
||||
|
||||
static Integer parseIntSafe(String text) {
|
||||
if (text == null) {
|
||||
return null;
|
||||
}
|
||||
String digitsOnly = text.trim().replace(",", "");
|
||||
if (!digitsOnly.matches("-?\\d+")) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
return Integer.parseInt(digitsOnly);
|
||||
} catch (NumberFormatException e) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* 본문 컨테이너에서 원본 게시물의 형태(문단, 굵게, 링크, 이미지 위치)를 보존한 채로
|
||||
* 안전하게 걸러낸 HTML과 이미지 URL 목록을 뽑아낸다.
|
||||
* img/a의 상대경로는 절대경로로 바꾼 뒤 Safelist로 스크립트 등 위험 요소만 제거한다.
|
||||
*/
|
||||
static PostDetail extractDetail(Element body) {
|
||||
if (body == null) {
|
||||
return PostDetail.EMPTY;
|
||||
}
|
||||
|
||||
Element working = body.clone();
|
||||
|
||||
Set<String> imageUrls = new LinkedHashSet<>();
|
||||
for (Element img : working.select("img")) {
|
||||
String src = img.attr("abs:src");
|
||||
if (src.isBlank()) {
|
||||
img.remove();
|
||||
continue;
|
||||
}
|
||||
img.attr("src", src);
|
||||
img.attr("referrerpolicy", "no-referrer");
|
||||
img.attr("loading", "lazy");
|
||||
imageUrls.add(src);
|
||||
}
|
||||
for (Element link : working.select("a")) {
|
||||
String href = link.attr("abs:href");
|
||||
if (href.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
link.attr("href", href);
|
||||
link.attr("target", "_blank");
|
||||
link.attr("rel", "noopener noreferrer");
|
||||
}
|
||||
|
||||
String contentHtml = Jsoup.clean(working.html(), working.baseUri(), CONTENT_SAFELIST).trim();
|
||||
|
||||
return new PostDetail(contentHtml, List.copyOf(imageUrls));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,46 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import lombok.RequiredArgsConstructor;
|
||||
import lombok.extern.slf4j.Slf4j;
|
||||
import org.springframework.scheduling.annotation.Scheduled;
|
||||
import org.springframework.stereotype.Component;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
@Slf4j
|
||||
@Component
|
||||
@RequiredArgsConstructor
|
||||
public class ScrapingScheduler {
|
||||
|
||||
// 사이트에 부담을 주지 않기 위해 게시판별 요청 사이 최소 간격을 둔다.
|
||||
private static final long DELAY_BETWEEN_BOARDS_MS = 1500;
|
||||
|
||||
private final BoardScraperService boardScraperService;
|
||||
private final ScrapedPostSyncService scrapedPostSyncService;
|
||||
|
||||
@Scheduled(
|
||||
initialDelayString = "${scraping.initial-delay-ms:10000}",
|
||||
fixedDelayString = "${scraping.interval-ms:600000}"
|
||||
)
|
||||
public void scrapeAllBoards() {
|
||||
for (BoardSource source : BoardSource.values()) {
|
||||
try {
|
||||
List<ScrapedPostDto> posts = boardScraperService.scrape(source);
|
||||
int savedCount = scrapedPostSyncService.saveNewPosts(source, posts);
|
||||
log.info("[스크래핑] {}/{} - 수집 {}건, 신규 저장 {}건",
|
||||
source.getSiteName(), source.getBoardName(), posts.size(), savedCount);
|
||||
} catch (Exception e) {
|
||||
log.warn("[스크래핑 실패] {}/{} - {}", source.getSiteName(), source.getBoardName(), e.getMessage());
|
||||
}
|
||||
sleepQuietly();
|
||||
}
|
||||
}
|
||||
|
||||
private void sleepQuietly() {
|
||||
try {
|
||||
Thread.sleep(DELAY_BETWEEN_BOARDS_MS);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -21,4 +21,11 @@ jwt:
|
||||
# 256비트 이상이어야 하므로 영문+숫자 (최소 32글자 이상)
|
||||
secret: c3ByaW5nYm9vdC1qd3Qtc2VjcmV0LWtleS1zcGVjaWFsc291cmNlLWhvbWVwYWdlLWJhY2tlbmQtMjAyNg==
|
||||
# 토큰 유효 시간 (1시간으로 설정 = 60분 * 60초 * 1000밀리초)
|
||||
expiration-time: 3600000
|
||||
expiration-time: 3600000
|
||||
|
||||
scraping:
|
||||
initial-delay-ms: 10000 # 앱 기동 후 첫 스크래핑까지 대기 시간
|
||||
interval-ms: 600000 # 게시판 전체 스크래핑 주기 (10분)
|
||||
retention-days: 7 # 이 기간(scraped_at 기준)이 지난 게시물은 자동 삭제
|
||||
cleanup-initial-delay-ms: 60000 # 앱 기동 후 첫 정리 작업까지 대기 시간
|
||||
cleanup-interval-ms: 86400000 # 정리 작업 주기 (24시간)
|
||||
Reference in New Issue
Block a user