Compare commits

...
3 Commits
Author SHA1 Message Date
headporter d7547a1028 chore : deploy 수정
SpecialSource Backend CI/CD / build-and-deploy (push) Successful in 1m5s
2026-08-12 10:56:41 +09:00
headporter 53f28e2f09 Merge remote-tracking branch 'origin/master'
SpecialSource Backend CI/CD / build-and-deploy (push) Successful in 1m25s
# Conflicts:
#	src/main/java/company/specialsource/SpecialsourceHomepageApplication.java
2026-08-11 17:58:12 +09:00
headporter fd26a086f7 chore : deploy 수정 2026-08-11 17:55:45 +09:00
40 changed files with 23277 additions and 3 deletions
+3
View File
@@ -51,6 +51,9 @@ dependencies {
runtimeOnly 'io.jsonwebtoken:jjwt-impl:0.12.5' runtimeOnly 'io.jsonwebtoken:jjwt-impl:0.12.5'
runtimeOnly 'io.jsonwebtoken:jjwt-jackson:0.12.5' // JSON 파싱용 runtimeOnly 'io.jsonwebtoken:jjwt-jackson:0.12.5' // JSON 파싱용
// 게시판 스크래핑용 HTML 파서
implementation 'org.jsoup:jsoup:1.18.1'
} }
tasks.named('test') { tasks.named('test') {
@@ -21,7 +21,7 @@ public class DataInitializer implements CommandLineRunner {
if (userRepository.findByUsername("admin").isEmpty()) { if (userRepository.findByUsername("admin").isEmpty()) {
userRepository.save(User.builder() userRepository.save(User.builder()
.username("admin") .username("admin")
.password(passwordEncoder.encode("1234")) .password(passwordEncoder.encode("Ppasse444$"))
.role("ROLE_USER") .role("ROLE_USER")
.build()); .build());
log.info("초기 관리자 계정(admin)이 생성되었습니다."); log.info("초기 관리자 계정(admin)이 생성되었습니다.");
@@ -0,0 +1,46 @@
package company.specialsource.config;
import jakarta.servlet.FilterChain;
import jakarta.servlet.ServletException;
import jakarta.servlet.http.HttpServletRequest;
import jakarta.servlet.http.HttpServletResponse;
import lombok.RequiredArgsConstructor;
import org.springframework.security.authentication.UsernamePasswordAuthenticationToken;
import org.springframework.security.core.context.SecurityContextHolder;
import org.springframework.stereotype.Component;
import org.springframework.web.filter.OncePerRequestFilter;
import java.io.IOException;
import java.util.List;
@Component
@RequiredArgsConstructor
public class JwtAuthenticationFilter extends OncePerRequestFilter {
private static final String HEADER_PREFIX = "Bearer ";
private final JwtTokenProvider jwtTokenProvider;
@Override
protected void doFilterInternal(
HttpServletRequest request,
HttpServletResponse response,
FilterChain filterChain
) throws ServletException, IOException {
String header = request.getHeader("Authorization");
if (header != null && header.startsWith(HEADER_PREFIX)) {
String token = header.substring(HEADER_PREFIX.length());
if (jwtTokenProvider.validateToken(token)) {
String username = jwtTokenProvider.getUsername(token);
UsernamePasswordAuthenticationToken authentication =
new UsernamePasswordAuthenticationToken(username, null, List.of());
SecurityContextHolder.getContext().setAuthentication(authentication);
}
}
filterChain.doFilter(request, response);
}
}
@@ -1,26 +1,34 @@
package company.specialsource.config; package company.specialsource.config;
import lombok.RequiredArgsConstructor;
import org.springframework.context.annotation.Bean; import org.springframework.context.annotation.Bean;
import org.springframework.context.annotation.Configuration; import org.springframework.context.annotation.Configuration;
import org.springframework.security.config.annotation.web.builders.HttpSecurity; import org.springframework.security.config.annotation.web.builders.HttpSecurity;
import org.springframework.security.config.annotation.web.configuration.EnableWebSecurity; import org.springframework.security.config.annotation.web.configuration.EnableWebSecurity;
import org.springframework.security.config.http.SessionCreationPolicy;
import org.springframework.security.crypto.bcrypt.BCryptPasswordEncoder; import org.springframework.security.crypto.bcrypt.BCryptPasswordEncoder;
import org.springframework.security.crypto.password.PasswordEncoder; import org.springframework.security.crypto.password.PasswordEncoder;
import org.springframework.security.web.SecurityFilterChain; import org.springframework.security.web.SecurityFilterChain;
import org.springframework.security.web.authentication.UsernamePasswordAuthenticationFilter;
@Configuration @Configuration
@EnableWebSecurity @EnableWebSecurity
@RequiredArgsConstructor
public class SecurityConfig { public class SecurityConfig {
private final JwtAuthenticationFilter jwtAuthenticationFilter;
@Bean @Bean
public SecurityFilterChain filterChain(HttpSecurity http) throws Exception { public SecurityFilterChain filterChain(HttpSecurity http) throws Exception {
http http
.csrf(csrf -> csrf.disable()) // REST API 통신을 위해 CSRF 비활성화 .csrf(csrf -> csrf.disable()) // REST API 통신을 위해 CSRF 비활성화
.sessionManagement(session -> session.sessionCreationPolicy(SessionCreationPolicy.STATELESS))
.authorizeHttpRequests(auth -> auth .authorizeHttpRequests(auth -> auth
// 로그인 API 및 파일 동기화 업로드 API는 인증 없이 누구나 접근 가능하도록 허용 // 로그인 API 및 파일 동기화 업로드 API는 인증 없이 누구나 접근 가능하도록 허용
.requestMatchers("/api/login", "/api/sync/upload").permitAll() .requestMatchers("/api/login", "/api/sync/upload").permitAll()
.anyRequest().authenticated() .anyRequest().authenticated()
); )
.addFilterBefore(jwtAuthenticationFilter, UsernamePasswordAuthenticationFilter.class);
return http.build(); return http.build();
} }
@@ -0,0 +1,71 @@
package company.specialsource.controller;
import company.specialsource.scraper.BoardSource;
import company.specialsource.scraper.BoardSummaryResponse;
import company.specialsource.scraper.ScrapedPostQueryService;
import company.specialsource.scraper.ScrapedPostResponse;
import lombok.RequiredArgsConstructor;
import org.springframework.data.domain.Page;
import org.springframework.data.domain.PageRequest;
import org.springframework.data.domain.Pageable;
import org.springframework.http.ResponseEntity;
import org.springframework.web.bind.annotation.*;
import java.util.Arrays;
import java.util.List;
@RestController
@RequestMapping("/api/scraped-posts")
@RequiredArgsConstructor
public class ScrapedPostController {
private final ScrapedPostQueryService scrapedPostQueryService;
@GetMapping
public ResponseEntity<?> getPosts(
@RequestParam(required = false) String board,
@RequestParam(defaultValue = "0") int page,
@RequestParam(defaultValue = "20") int size
) {
if (board != null && !isValidBoardKey(board)) {
return ResponseEntity.badRequest().body("존재하지 않는 게시판 키입니다: " + board);
}
Pageable pageable = PageRequest.of(page, Math.min(size, 100));
Page<ScrapedPostResponse> posts = scrapedPostQueryService.getPosts(board, pageable);
return ResponseEntity.ok(posts);
}
@GetMapping("/viewed")
public ResponseEntity<?> getViewedPosts(
@RequestParam(required = false) String board,
@RequestParam(defaultValue = "0") int page,
@RequestParam(defaultValue = "20") int size
) {
if (board != null && !isValidBoardKey(board)) {
return ResponseEntity.badRequest().body("존재하지 않는 게시판 키입니다: " + board);
}
Pageable pageable = PageRequest.of(page, Math.min(size, 100));
Page<ScrapedPostResponse> posts = scrapedPostQueryService.getViewedPosts(board, pageable);
return ResponseEntity.ok(posts);
}
@GetMapping("/boards")
public ResponseEntity<List<BoardSummaryResponse>> getBoards() {
List<BoardSummaryResponse> boards = Arrays.stream(BoardSource.values())
.map(BoardSummaryResponse::from)
.toList();
return ResponseEntity.ok(boards);
}
@PostMapping("/{id}/view")
public ResponseEntity<Void> markViewed(@PathVariable Long id) {
scrapedPostQueryService.markViewed(id);
return ResponseEntity.noContent().build();
}
private boolean isValidBoardKey(String boardKey) {
return Arrays.stream(BoardSource.values()).anyMatch(source -> source.name().equals(boardKey));
}
}
@@ -0,0 +1,56 @@
package company.specialsource.entity;
import jakarta.persistence.*;
import lombok.*;
import java.time.LocalDateTime;
@Entity
@Table(name = "scraped_posts",
uniqueConstraints = @UniqueConstraint(columnNames = {"board_key", "link"}))
@Getter
@NoArgsConstructor(access = AccessLevel.PROTECTED)
@AllArgsConstructor
@Builder
public class ScrapedPost {
@Id
@GeneratedValue(strategy = GenerationType.IDENTITY)
private Long id;
@Column(name = "board_key", nullable = false)
private String boardKey; // BoardSource enum name, 예: RULIWEB_BEST
@Column(name = "site_name", nullable = false)
private String siteName; // 예: 루리웹, 디씨인사이드
@Column(name = "board_name", nullable = false)
private String boardName; // 예: 실시간베스트, 주짓수 갤러리
@Column(nullable = false, length = 500)
private String title;
@Column(nullable = false, length = 1000)
private String link;
private String author;
private String postedAt; // 사이트마다 표기 형식이 달라(HH:mm, yy.MM.dd 등) 원문 그대로 저장
private Integer viewCount;
private Integer likeCount;
@Column(columnDefinition = "TEXT")
private String content; // 게시물 상세페이지에서 추출한 본문 (정제된 HTML, 원본 문단/링크/이미지 위치 보존)
@Column(name = "image_urls", columnDefinition = "TEXT")
private String imageUrls; // 본문에 포함된 이미지 URL, 줄바꿈으로 구분
@Column(name = "scraped_at", nullable = false)
private LocalDateTime scrapedAt;
@Setter
@Column(name = "viewed_at")
private LocalDateTime viewedAt; // 게시판 조회 시각. null이면 아직 안 읽은 글
}
@@ -0,0 +1,24 @@
package company.specialsource.repository;
import company.specialsource.entity.ScrapedPost;
import org.springframework.data.domain.Page;
import org.springframework.data.domain.Pageable;
import org.springframework.data.jpa.repository.JpaRepository;
import org.springframework.transaction.annotation.Transactional;
import java.time.LocalDateTime;
public interface ScrapedPostRepository extends JpaRepository<ScrapedPost, Long> {
boolean existsByBoardKeyAndLink(String boardKey, String link);
Page<ScrapedPost> findByBoardKeyAndViewedAtIsNullOrderByScrapedAtDesc(String boardKey, Pageable pageable);
Page<ScrapedPost> findByViewedAtIsNullOrderByScrapedAtDesc(Pageable pageable);
Page<ScrapedPost> findByBoardKeyAndViewedAtIsNotNullOrderByViewedAtDesc(String boardKey, Pageable pageable);
Page<ScrapedPost> findByViewedAtIsNotNullOrderByViewedAtDesc(Pageable pageable);
@Transactional
long deleteByScrapedAtBefore(LocalDateTime cutoff);
}
@@ -0,0 +1,9 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import java.util.List;
public interface BoardParser {
List<ScrapedPostDto> parse(Document document);
}
@@ -0,0 +1,37 @@
package company.specialsource.scraper;
import lombok.RequiredArgsConstructor;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.springframework.stereotype.Service;
import java.io.IOException;
import java.util.List;
@Service
@RequiredArgsConstructor
public class BoardScraperService {
private static final String USER_AGENT =
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36";
private static final int TIMEOUT_MS = 10_000;
private final RuliwebBoardListParser ruliwebBoardListParser;
private final RuliwebMypiParser ruliwebMypiParser;
private final DcGalleryParser dcGalleryParser;
public List<ScrapedPostDto> scrape(BoardSource source) throws IOException {
Document document = Jsoup.connect(source.getUrl())
.userAgent(USER_AGENT)
.timeout(TIMEOUT_MS)
.get();
BoardParser parser = switch (source.getParserType()) {
case RULIWEB_BOARD_LIST -> ruliwebBoardListParser;
case RULIWEB_MYPI_BEST -> ruliwebMypiParser;
case DC_GALLERY_LIST -> dcGalleryParser;
};
return parser.parse(document);
}
}
@@ -0,0 +1,27 @@
package company.specialsource.scraper;
import lombok.Getter;
@Getter
public enum BoardSource {
RULIWEB_BEST("루리웹", "실시간베스트", "https://bbs.ruliweb.com/best/selection", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
RULIWEB_HUMOR("루리웹", "유머", "https://bbs.ruliweb.com/best/humor", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
RULIWEB_MYPI("루리웹", "마이피", "https://mypi.ruliweb.com/", ParserType.RULIWEB_MYPI_BEST, DetailParserType.RULIWEB),
RULIWEB_HOTDEAL("루리웹", "핫딜", "https://bbs.ruliweb.com/market/board/1020", ParserType.RULIWEB_BOARD_LIST, DetailParserType.RULIWEB),
DC_BJJ("디씨인사이드", "주짓수 갤러리", "https://gall.dcinside.com/mgallery/board/lists/?id=bjj", ParserType.DC_GALLERY_LIST, DetailParserType.DC),
DC_SINGULARITY("디씨인사이드", "특이점이 온다", "https://gall.dcinside.com/mgallery/board/lists?id=thesingularity", ParserType.DC_GALLERY_LIST, DetailParserType.DC);
private final String siteName;
private final String boardName;
private final String url;
private final ParserType parserType;
private final DetailParserType detailParserType;
BoardSource(String siteName, String boardName, String url, ParserType parserType, DetailParserType detailParserType) {
this.siteName = siteName;
this.boardName = boardName;
this.url = url;
this.parserType = parserType;
this.detailParserType = detailParserType;
}
}
@@ -0,0 +1,7 @@
package company.specialsource.scraper;
public record BoardSummaryResponse(String boardKey, String siteName, String boardName) {
public static BoardSummaryResponse from(BoardSource source) {
return new BoardSummaryResponse(source.name(), source.getSiteName(), source.getBoardName());
}
}
@@ -0,0 +1,53 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import org.jsoup.nodes.Element;
import org.springframework.stereotype.Component;
import java.util.ArrayList;
import java.util.List;
/**
* 디씨인사이드 갤러리 목록 파서. 공지/설문 행은 제외하고 일반 게시글만 수집한다.
*/
@Component
public class DcGalleryParser implements BoardParser {
@Override
public List<ScrapedPostDto> parse(Document document) {
List<ScrapedPostDto> posts = new ArrayList<>();
for (Element row : document.select("tr.ub-content")) {
if (!row.hasClass("us-post") || "icon_notice".equals(row.attr("data-type"))) {
continue;
}
Element titleAnchor = row.selectFirst("td.gall_tit a");
if (titleAnchor == null) {
continue;
}
String title = titleAnchor.text().trim();
String link = titleAnchor.attr("abs:href");
if (title.isEmpty() || link.isEmpty()) {
continue;
}
Element writerEl = row.selectFirst("td.gall_writer");
String author = writerEl == null ? "" : writerEl.attr("data-nick");
if (author.isBlank() && writerEl != null) {
author = writerEl.text().trim();
}
Element dateEl = row.selectFirst("td.gall_date");
String postedAt = dateEl == null ? "" : dateEl.hasAttr("title") ? dateEl.attr("title") : dateEl.text().trim();
Integer viewCount = ScraperTextUtils.parseIntSafe(row.select("td.gall_count").text());
Integer likeCount = ScraperTextUtils.parseIntSafe(row.select("td.gall_recommend").text());
posts.add(new ScrapedPostDto(title, link, author, postedAt, viewCount, likeCount));
}
return posts;
}
}
@@ -0,0 +1,13 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import org.springframework.stereotype.Component;
@Component
public class DcPostDetailParser implements DetailParser {
@Override
public PostDetail parse(Document document) {
return ScraperTextUtils.extractDetail(document.selectFirst("div.write_div"));
}
}
@@ -0,0 +1,7 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
public interface DetailParser {
PostDetail parse(Document document);
}
@@ -0,0 +1,6 @@
package company.specialsource.scraper;
public enum DetailParserType {
RULIWEB,
DC
}
@@ -0,0 +1,7 @@
package company.specialsource.scraper;
public enum ParserType {
RULIWEB_BOARD_LIST, // 표 형태 게시판 (실시간베스트, 유머, 핫딜)
RULIWEB_MYPI_BEST, // 마이피 실시간 인기글 위젯
DC_GALLERY_LIST // 디씨인사이드 갤러리 게시판
}
@@ -0,0 +1,10 @@
package company.specialsource.scraper;
import java.util.List;
/**
* content는 원본 게시물의 문단/굵게/링크/이미지 위치를 보존한 정제된(sanitized) HTML이다.
*/
public record PostDetail(String content, List<String> imageUrls) {
static final PostDetail EMPTY = new PostDetail("", List.of());
}
@@ -0,0 +1,34 @@
package company.specialsource.scraper;
import lombok.RequiredArgsConstructor;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.springframework.stereotype.Service;
import java.io.IOException;
@Service
@RequiredArgsConstructor
public class PostDetailScraperService {
private static final String USER_AGENT =
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36";
private static final int TIMEOUT_MS = 10_000;
private final RuliwebPostDetailParser ruliwebPostDetailParser;
private final DcPostDetailParser dcPostDetailParser;
public PostDetail scrape(String url, DetailParserType type) throws IOException {
Document document = Jsoup.connect(url)
.userAgent(USER_AGENT)
.timeout(TIMEOUT_MS)
.get();
DetailParser parser = switch (type) {
case RULIWEB -> ruliwebPostDetailParser;
case DC -> dcPostDetailParser;
};
return parser.parse(document);
}
}
@@ -0,0 +1,48 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import org.jsoup.nodes.Element;
import org.springframework.stereotype.Component;
import java.util.ArrayList;
import java.util.List;
/**
* 루리웹 표 형태 게시판(실시간베스트, 유머, 핫딜) 공통 파서.
* 게시판마다 tr.table_body 구조는 같고, 공지(class=notice) 행만 다르게 섞여 있어 걸러낸다.
*/
@Component
public class RuliwebBoardListParser implements BoardParser {
@Override
public List<ScrapedPostDto> parse(Document document) {
List<ScrapedPostDto> posts = new ArrayList<>();
for (Element row : document.select("tr.table_body")) {
if (row.hasClass("notice")) {
continue;
}
Element titleAnchor = row.selectFirst("td.subject a");
if (titleAnchor == null) {
continue;
}
Element strong = titleAnchor.selectFirst("strong");
String title = (strong != null ? strong.text() : titleAnchor.text()).trim();
String link = titleAnchor.attr("abs:href");
if (title.isEmpty() || link.isEmpty()) {
continue;
}
String author = row.select("td.writer").text().trim();
String postedAt = row.select("td.time").text().trim();
Integer likeCount = ScraperTextUtils.parseIntSafe(row.select("td.recomd").text());
Integer viewCount = ScraperTextUtils.parseIntSafe(row.select("td.hit").text());
posts.add(new ScrapedPostDto(title, link, author, postedAt, viewCount, likeCount));
}
return posts;
}
}
@@ -0,0 +1,36 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import org.jsoup.nodes.Element;
import org.springframework.stereotype.Component;
import java.util.ArrayList;
import java.util.List;
/**
* 마이피(mypi.ruliweb.com)의 "실시간 인기글" 위젯 파서.
* 작성자/날짜/조회수는 목록에 노출되지 않아 title, link만 수집한다.
*/
@Component
public class RuliwebMypiParser implements BoardParser {
@Override
public List<ScrapedPostDto> parse(Document document) {
List<ScrapedPostDto> posts = new ArrayList<>();
for (Element anchor : document.select("ul.right_best_list li.right_best_list_item a.txt_link")) {
String title = anchor.text().trim();
String link = anchor.attr("abs:href");
if (title.isEmpty() || link.isEmpty()) {
continue;
}
// 위젯 목록에 광고(외부 리다이렉트 링크)가 섞여 들어오는 경우가 있어 루리웹 도메인만 통과시킨다.
if (!link.contains("ruliweb.com")) {
continue;
}
posts.add(new ScrapedPostDto(title, link, null, null, null, null));
}
return posts;
}
}
@@ -0,0 +1,13 @@
package company.specialsource.scraper;
import org.jsoup.nodes.Document;
import org.springframework.stereotype.Component;
@Component
public class RuliwebPostDetailParser implements DetailParser {
@Override
public PostDetail parse(Document document) {
return ScraperTextUtils.extractDetail(document.selectFirst("div.view_content"));
}
}
@@ -0,0 +1,37 @@
package company.specialsource.scraper;
import company.specialsource.repository.ScrapedPostRepository;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.beans.factory.annotation.Value;
import org.springframework.scheduling.annotation.Scheduled;
import org.springframework.stereotype.Component;
import java.time.LocalDateTime;
/**
* 오래된 스크랩 게시물을 주기적으로 정리한다.
* 스크래핑으로 계속 쌓이기만 하는 캐시성 데이터라 일정 기간이 지나면 삭제해도 무방하다.
*/
@Slf4j
@Component
@RequiredArgsConstructor
public class ScrapedPostCleanupScheduler {
private final ScrapedPostRepository scrapedPostRepository;
@Value("${scraping.retention-days:7}")
private int retentionDays;
@Scheduled(
initialDelayString = "${scraping.cleanup-initial-delay-ms:60000}",
fixedDelayString = "${scraping.cleanup-interval-ms:86400000}"
)
public void deleteExpiredPosts() {
LocalDateTime cutoff = LocalDateTime.now().minusDays(retentionDays);
long deletedCount = scrapedPostRepository.deleteByScrapedAtBefore(cutoff);
if (deletedCount > 0) {
log.info("[정리] 스크랩 {}일 경과 게시물 {}건 삭제 (기준: {})", retentionDays, deletedCount, cutoff);
}
}
}
@@ -0,0 +1,11 @@
package company.specialsource.scraper;
public record ScrapedPostDto(
String title,
String link,
String author,
String postedAt,
Integer viewCount,
Integer likeCount
) {
}
@@ -0,0 +1,44 @@
package company.specialsource.scraper;
import company.specialsource.entity.ScrapedPost;
import company.specialsource.repository.ScrapedPostRepository;
import jakarta.persistence.EntityNotFoundException;
import lombok.RequiredArgsConstructor;
import org.springframework.data.domain.Page;
import org.springframework.data.domain.Pageable;
import org.springframework.stereotype.Service;
import org.springframework.transaction.annotation.Transactional;
import java.time.LocalDateTime;
@Service
@RequiredArgsConstructor
public class ScrapedPostQueryService {
private final ScrapedPostRepository scrapedPostRepository;
public Page<ScrapedPostResponse> getPosts(String boardKey, Pageable pageable) {
// 이미 조회한(viewedAt != null) 글은 목록에서 제외한다.
Page<ScrapedPost> page = (boardKey == null || boardKey.isBlank())
? scrapedPostRepository.findByViewedAtIsNullOrderByScrapedAtDesc(pageable)
: scrapedPostRepository.findByBoardKeyAndViewedAtIsNullOrderByScrapedAtDesc(boardKey, pageable);
return page.map(ScrapedPostResponse::from);
}
public Page<ScrapedPostResponse> getViewedPosts(String boardKey, Pageable pageable) {
// 조회한(viewedAt != null) 글만 최근 읽은 순으로 보여준다.
Page<ScrapedPost> page = (boardKey == null || boardKey.isBlank())
? scrapedPostRepository.findByViewedAtIsNotNullOrderByViewedAtDesc(pageable)
: scrapedPostRepository.findByBoardKeyAndViewedAtIsNotNullOrderByViewedAtDesc(boardKey, pageable);
return page.map(ScrapedPostResponse::from);
}
@Transactional
public void markViewed(Long id) {
ScrapedPost post = scrapedPostRepository.findById(id)
.orElseThrow(() -> new EntityNotFoundException("게시물을 찾을 수 없습니다: " + id));
post.setViewedAt(LocalDateTime.now());
}
}
@@ -0,0 +1,46 @@
package company.specialsource.scraper;
import company.specialsource.entity.ScrapedPost;
import java.time.LocalDateTime;
import java.util.List;
public record ScrapedPostResponse(
Long id,
String boardKey,
String siteName,
String boardName,
String title,
String link,
String author,
String postedAt,
Integer viewCount,
Integer likeCount,
String content,
List<String> imageUrls,
LocalDateTime scrapedAt,
LocalDateTime viewedAt
) {
public static ScrapedPostResponse from(ScrapedPost post) {
List<String> imageUrls = (post.getImageUrls() == null || post.getImageUrls().isBlank())
? List.of()
: List.of(post.getImageUrls().split("\n"));
return new ScrapedPostResponse(
post.getId(),
post.getBoardKey(),
post.getSiteName(),
post.getBoardName(),
post.getTitle(),
post.getLink(),
post.getAuthor(),
post.getPostedAt(),
post.getViewCount(),
post.getLikeCount(),
post.getContent(),
imageUrls,
post.getScrapedAt(),
post.getViewedAt()
);
}
}
@@ -0,0 +1,72 @@
package company.specialsource.scraper;
import company.specialsource.entity.ScrapedPost;
import company.specialsource.repository.ScrapedPostRepository;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.stereotype.Service;
import java.time.LocalDateTime;
import java.util.List;
/**
* 목록에 새로 나타난 글만 상세페이지까지 들어가서 본문을 수집한다.
* 이미 저장된 글은 상세 요청을 다시 하지 않으므로 매 주기마다 요청량이 쌓이지 않는다.
*/
@Slf4j
@Service
@RequiredArgsConstructor
public class ScrapedPostSyncService {
private static final long DELAY_BETWEEN_DETAIL_FETCHES_MS = 800;
private final ScrapedPostRepository scrapedPostRepository;
private final PostDetailScraperService postDetailScraperService;
public int saveNewPosts(BoardSource source, List<ScrapedPostDto> posts) {
int savedCount = 0;
for (ScrapedPostDto post : posts) {
if (scrapedPostRepository.existsByBoardKeyAndLink(source.name(), post.link())) {
continue;
}
PostDetail detail = fetchDetailSafely(source, post);
scrapedPostRepository.save(ScrapedPost.builder()
.boardKey(source.name())
.siteName(source.getSiteName())
.boardName(source.getBoardName())
.title(post.title())
.link(post.link())
.author(post.author())
.postedAt(post.postedAt())
.viewCount(post.viewCount())
.likeCount(post.likeCount())
.content(detail.content())
.imageUrls(String.join("\n", detail.imageUrls()))
.scrapedAt(LocalDateTime.now())
.build());
savedCount++;
sleepQuietly();
}
return savedCount;
}
private PostDetail fetchDetailSafely(BoardSource source, ScrapedPostDto post) {
try {
return postDetailScraperService.scrape(post.link(), source.getDetailParserType());
} catch (Exception e) {
log.warn("[본문 스크래핑 실패] {} - {}", post.link(), e.getMessage());
return PostDetail.EMPTY;
}
}
private void sleepQuietly() {
try {
Thread.sleep(DELAY_BETWEEN_DETAIL_FETCHES_MS);
} catch (InterruptedException e) {
Thread.currentThread().interrupt();
}
}
}
@@ -0,0 +1,73 @@
package company.specialsource.scraper;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Element;
import org.jsoup.safety.Safelist;
import java.util.LinkedHashSet;
import java.util.List;
import java.util.Set;
final class ScraperTextUtils {
private static final Safelist CONTENT_SAFELIST = Safelist.relaxed()
.addAttributes("img", "src", "alt", "referrerpolicy", "loading")
.addAttributes("a", "href", "target", "rel");
private ScraperTextUtils() {
}
static Integer parseIntSafe(String text) {
if (text == null) {
return null;
}
String digitsOnly = text.trim().replace(",", "");
if (!digitsOnly.matches("-?\\d+")) {
return null;
}
try {
return Integer.parseInt(digitsOnly);
} catch (NumberFormatException e) {
return null;
}
}
/**
* 본문 컨테이너에서 원본 게시물의 형태(문단, 굵게, 링크, 이미지 위치)를 보존한 채로
* 안전하게 걸러낸 HTML과 이미지 URL 목록을 뽑아낸다.
* img/a의 상대경로는 절대경로로 바꾼 뒤 Safelist로 스크립트 등 위험 요소만 제거한다.
*/
static PostDetail extractDetail(Element body) {
if (body == null) {
return PostDetail.EMPTY;
}
Element working = body.clone();
Set<String> imageUrls = new LinkedHashSet<>();
for (Element img : working.select("img")) {
String src = img.attr("abs:src");
if (src.isBlank()) {
img.remove();
continue;
}
img.attr("src", src);
img.attr("referrerpolicy", "no-referrer");
img.attr("loading", "lazy");
imageUrls.add(src);
}
for (Element link : working.select("a")) {
String href = link.attr("abs:href");
if (href.isBlank()) {
continue;
}
link.attr("href", href);
link.attr("target", "_blank");
link.attr("rel", "noopener noreferrer");
}
String contentHtml = Jsoup.clean(working.html(), working.baseUri(), CONTENT_SAFELIST).trim();
return new PostDetail(contentHtml, List.copyOf(imageUrls));
}
}
@@ -0,0 +1,46 @@
package company.specialsource.scraper;
import lombok.RequiredArgsConstructor;
import lombok.extern.slf4j.Slf4j;
import org.springframework.scheduling.annotation.Scheduled;
import org.springframework.stereotype.Component;
import java.util.List;
@Slf4j
@Component
@RequiredArgsConstructor
public class ScrapingScheduler {
// 사이트에 부담을 주지 않기 위해 게시판별 요청 사이 최소 간격을 둔다.
private static final long DELAY_BETWEEN_BOARDS_MS = 1500;
private final BoardScraperService boardScraperService;
private final ScrapedPostSyncService scrapedPostSyncService;
@Scheduled(
initialDelayString = "${scraping.initial-delay-ms:10000}",
fixedDelayString = "${scraping.interval-ms:600000}"
)
public void scrapeAllBoards() {
for (BoardSource source : BoardSource.values()) {
try {
List<ScrapedPostDto> posts = boardScraperService.scrape(source);
int savedCount = scrapedPostSyncService.saveNewPosts(source, posts);
log.info("[스크래핑] {}/{} - 수집 {}건, 신규 저장 {}건",
source.getSiteName(), source.getBoardName(), posts.size(), savedCount);
} catch (Exception e) {
log.warn("[스크래핑 실패] {}/{} - {}", source.getSiteName(), source.getBoardName(), e.getMessage());
}
sleepQuietly();
}
}
private void sleepQuietly() {
try {
Thread.sleep(DELAY_BETWEEN_BOARDS_MS);
} catch (InterruptedException e) {
Thread.currentThread().interrupt();
}
}
}
+8 -1
View File
@@ -36,4 +36,11 @@ jwt:
# 256비트 이상이어야 하므로 영문+숫자 (최소 32글자 이상) # 256비트 이상이어야 하므로 영문+숫자 (최소 32글자 이상)
secret: c3ByaW5nYm9vdC1qd3Qtc2VjcmV0LWtleS1zcGVjaWFsc291cmNlLWhvbWVwYWdlLWJhY2tlbmQtMjAyNg== secret: c3ByaW5nYm9vdC1qd3Qtc2VjcmV0LWtleS1zcGVjaWFsc291cmNlLWhvbWVwYWdlLWJhY2tlbmQtMjAyNg==
# 토큰 유효 시간 (1시간으로 설정 = 60분 * 60초 * 1000밀리초) # 토큰 유효 시간 (1시간으로 설정 = 60분 * 60초 * 1000밀리초)
expiration-time: 3600000 expiration-time: 3600000
scraping:
initial-delay-ms: 10000 # 앱 기동 후 첫 스크래핑까지 대기 시간
interval-ms: 600000 # 게시판 전체 스크래핑 주기 (10분)
retention-days: 7 # 이 기간(scraped_at 기준)이 지난 게시물은 자동 삭제
cleanup-initial-delay-ms: 60000 # 앱 기동 후 첫 정리 작업까지 대기 시간
cleanup-interval-ms: 86400000 # 정리 작업 주기 (24시간)
+8
View File
@@ -0,0 +1,8 @@
POST http://localhost:8080/api/login
#POST http://192.168.0.215:18080/api/login
Content-Type: application/json
{
"username": "admin",
"password": "1234"
}
@@ -0,0 +1,13 @@
package company.specialsource;
import org.junit.jupiter.api.Test;
import org.springframework.boot.test.context.SpringBootTest;
@SpringBootTest
class SpecialsourceHomepageApplicationTests {
@Test
void contextLoads() {
}
}
@@ -0,0 +1,37 @@
package company.specialsource.config;
import org.jasypt.encryption.StringEncryptor;
import org.junit.jupiter.api.Test;
import org.springframework.beans.factory.annotation.Qualifier;
import org.springframework.boot.test.context.SpringBootTest;
import static org.assertj.core.api.Assertions.assertThat;
@SpringBootTest
class JasyptConfigTest {
private final StringEncryptor stringEncryptor;
JasyptConfigTest(@Qualifier("jasyptStringEncryptor") StringEncryptor stringEncryptor) {
this.stringEncryptor = stringEncryptor;
}
@Test
void encryptDatasourcePassword() {
String plainPassword = "q1w2e3r4$";
String encryptedPassword = stringEncryptor.encrypt(plainPassword);
String decryptedPassword = stringEncryptor.decrypt(encryptedPassword);
System.out.println("암호화 전:");
System.out.println(plainPassword);
System.out.println("암호화 결과:");
System.out.println("ENC(" + encryptedPassword + ")");
System.out.println("복호화 확인:");
System.out.println(decryptedPassword);
assertThat(decryptedPassword).isEqualTo(plainPassword);
}
}
@@ -0,0 +1,82 @@
package company.specialsource.scraper;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.junit.jupiter.api.Test;
import java.io.IOException;
import java.io.InputStream;
import java.util.List;
import static org.assertj.core.api.Assertions.assertThat;
/**
* 실제 저장해둔 게시판 HTML 스냅샷으로 파서가 제목/링크를 제대로 뽑아내는지 검증한다.
* 네트워크 접속 없이 동작하므로 사이트 구조가 바뀌지 않는 한 항상 재현 가능하다.
*/
class BoardParserTest {
private Document loadFixture(String fileName) throws IOException {
try (InputStream in = getClass().getResourceAsStream("/scraper/" + fileName)) {
assertThat(in).as("fixture %s must exist", fileName).isNotNull();
return Jsoup.parse(in, "UTF-8", "https://bbs.ruliweb.com/");
}
}
@Test
void ruliwebBoardListParser_parsesBestBoard() throws IOException {
Document doc = loadFixture("ruliweb_best.html");
List<ScrapedPostDto> posts = new RuliwebBoardListParser().parse(doc);
assertThat(posts).isNotEmpty();
ScrapedPostDto first = posts.get(0);
assertThat(first.title()).isNotBlank();
assertThat(first.link()).startsWith("https://bbs.ruliweb.com/best/board/");
assertThat(first.author()).isNotBlank();
}
@Test
void ruliwebBoardListParser_excludesNoticeRowsOnMarketBoard() throws IOException {
Document doc = loadFixture("ruliweb_market.html");
List<ScrapedPostDto> posts = new RuliwebBoardListParser().parse(doc);
assertThat(posts).isNotEmpty();
assertThat(posts).noneMatch(p -> p.title().contains("불법촬영물등 유통 방지"));
}
@Test
void ruliwebMypiParser_parsesBestWidget() throws IOException {
Document doc = loadFixture("ruliweb_mypi.html");
List<ScrapedPostDto> posts = new RuliwebMypiParser().parse(doc);
assertThat(posts).isNotEmpty();
assertThat(posts.get(0).title()).isNotBlank();
assertThat(posts.get(0).link()).startsWith("http");
assertThat(posts).allMatch(p -> p.link().contains("ruliweb.com"));
}
@Test
void ruliwebMypiParser_excludesAdLinksMixedIntoWidget() throws IOException {
Document doc = loadFixture("ruliweb_mypi.html");
List<ScrapedPostDto> posts = new RuliwebMypiParser().parse(doc);
assertThat(posts).noneMatch(p -> p.title().contains("인터넷 가입 변경 지원금"));
assertThat(posts).noneMatch(p -> p.link().contains("dajooda.com"));
}
@Test
void dcGalleryParser_excludesNoticeAndSurveyRows() throws IOException {
Document doc = loadFixture("dc_bjj.html");
List<ScrapedPostDto> posts = new DcGalleryParser().parse(doc);
assertThat(posts).isNotEmpty();
assertThat(posts).noneMatch(p -> p.title().contains("주짓수 갤러리 가이드"));
assertThat(posts).noneMatch(p -> p.title().contains("말빨로 정치인도"));
assertThat(posts.get(0).author()).isNotBlank();
}
}
@@ -0,0 +1,44 @@
package company.specialsource.scraper;
import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.junit.jupiter.api.Test;
import java.io.IOException;
import java.io.InputStream;
import static org.assertj.core.api.Assertions.assertThat;
class DetailParserTest {
private Document loadFixture(String fileName) throws IOException {
try (InputStream in = getClass().getResourceAsStream("/scraper/" + fileName)) {
assertThat(in).as("fixture %s must exist", fileName).isNotNull();
return Jsoup.parse(in, "UTF-8", "https://bbs.ruliweb.com/");
}
}
@Test
void ruliwebPostDetailParser_extractsTextAndImages() throws IOException {
Document doc = loadFixture("ruliweb_post_market.html");
PostDetail detail = new RuliwebPostDetailParser().parse(doc);
assertThat(Jsoup.parse(detail.content()).text()).contains("핫딜관리자입니다");
assertThat(detail.content()).contains("<img");
assertThat(detail.imageUrls()).isNotEmpty();
assertThat(detail.imageUrls().get(0)).startsWith("https://");
}
@Test
void dcPostDetailParser_extractsTextAndImages() throws IOException {
Document doc = loadFixture("dc_post.html");
PostDetail detail = new DcPostDetailParser().parse(doc);
assertThat(Jsoup.parse(detail.content()).text()).contains("출퇴근 소요 시간");
assertThat(detail.content()).contains("<img");
assertThat(detail.imageUrls()).isNotEmpty();
assertThat(detail.imageUrls().get(0)).contains("dcinside.co.kr");
}
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
File diff suppressed because one or more lines are too long