mirror of
https://github.com/headporter81/specialsource-homepage-backend.git
synced 2026-10-07 15:21:09 +09:00
chore : deploy 수정
This commit is contained in:
@@ -0,0 +1,82 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.Jsoup;
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
import java.util.List;
|
||||
|
||||
import static org.assertj.core.api.Assertions.assertThat;
|
||||
|
||||
/**
|
||||
* 실제 저장해둔 게시판 HTML 스냅샷으로 파서가 제목/링크를 제대로 뽑아내는지 검증한다.
|
||||
* 네트워크 접속 없이 동작하므로 사이트 구조가 바뀌지 않는 한 항상 재현 가능하다.
|
||||
*/
|
||||
class BoardParserTest {
|
||||
|
||||
private Document loadFixture(String fileName) throws IOException {
|
||||
try (InputStream in = getClass().getResourceAsStream("/scraper/" + fileName)) {
|
||||
assertThat(in).as("fixture %s must exist", fileName).isNotNull();
|
||||
return Jsoup.parse(in, "UTF-8", "https://bbs.ruliweb.com/");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void ruliwebBoardListParser_parsesBestBoard() throws IOException {
|
||||
Document doc = loadFixture("ruliweb_best.html");
|
||||
|
||||
List<ScrapedPostDto> posts = new RuliwebBoardListParser().parse(doc);
|
||||
|
||||
assertThat(posts).isNotEmpty();
|
||||
ScrapedPostDto first = posts.get(0);
|
||||
assertThat(first.title()).isNotBlank();
|
||||
assertThat(first.link()).startsWith("https://bbs.ruliweb.com/best/board/");
|
||||
assertThat(first.author()).isNotBlank();
|
||||
}
|
||||
|
||||
@Test
|
||||
void ruliwebBoardListParser_excludesNoticeRowsOnMarketBoard() throws IOException {
|
||||
Document doc = loadFixture("ruliweb_market.html");
|
||||
|
||||
List<ScrapedPostDto> posts = new RuliwebBoardListParser().parse(doc);
|
||||
|
||||
assertThat(posts).isNotEmpty();
|
||||
assertThat(posts).noneMatch(p -> p.title().contains("불법촬영물등 유통 방지"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void ruliwebMypiParser_parsesBestWidget() throws IOException {
|
||||
Document doc = loadFixture("ruliweb_mypi.html");
|
||||
|
||||
List<ScrapedPostDto> posts = new RuliwebMypiParser().parse(doc);
|
||||
|
||||
assertThat(posts).isNotEmpty();
|
||||
assertThat(posts.get(0).title()).isNotBlank();
|
||||
assertThat(posts.get(0).link()).startsWith("http");
|
||||
assertThat(posts).allMatch(p -> p.link().contains("ruliweb.com"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void ruliwebMypiParser_excludesAdLinksMixedIntoWidget() throws IOException {
|
||||
Document doc = loadFixture("ruliweb_mypi.html");
|
||||
|
||||
List<ScrapedPostDto> posts = new RuliwebMypiParser().parse(doc);
|
||||
|
||||
assertThat(posts).noneMatch(p -> p.title().contains("인터넷 가입 변경 지원금"));
|
||||
assertThat(posts).noneMatch(p -> p.link().contains("dajooda.com"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void dcGalleryParser_excludesNoticeAndSurveyRows() throws IOException {
|
||||
Document doc = loadFixture("dc_bjj.html");
|
||||
|
||||
List<ScrapedPostDto> posts = new DcGalleryParser().parse(doc);
|
||||
|
||||
assertThat(posts).isNotEmpty();
|
||||
assertThat(posts).noneMatch(p -> p.title().contains("주짓수 갤러리 가이드"));
|
||||
assertThat(posts).noneMatch(p -> p.title().contains("말빨로 정치인도"));
|
||||
assertThat(posts.get(0).author()).isNotBlank();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
package company.specialsource.scraper;
|
||||
|
||||
import org.jsoup.Jsoup;
|
||||
import org.jsoup.nodes.Document;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.InputStream;
|
||||
|
||||
import static org.assertj.core.api.Assertions.assertThat;
|
||||
|
||||
class DetailParserTest {
|
||||
|
||||
private Document loadFixture(String fileName) throws IOException {
|
||||
try (InputStream in = getClass().getResourceAsStream("/scraper/" + fileName)) {
|
||||
assertThat(in).as("fixture %s must exist", fileName).isNotNull();
|
||||
return Jsoup.parse(in, "UTF-8", "https://bbs.ruliweb.com/");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void ruliwebPostDetailParser_extractsTextAndImages() throws IOException {
|
||||
Document doc = loadFixture("ruliweb_post_market.html");
|
||||
|
||||
PostDetail detail = new RuliwebPostDetailParser().parse(doc);
|
||||
|
||||
assertThat(Jsoup.parse(detail.content()).text()).contains("핫딜관리자입니다");
|
||||
assertThat(detail.content()).contains("<img");
|
||||
assertThat(detail.imageUrls()).isNotEmpty();
|
||||
assertThat(detail.imageUrls().get(0)).startsWith("https://");
|
||||
}
|
||||
|
||||
@Test
|
||||
void dcPostDetailParser_extractsTextAndImages() throws IOException {
|
||||
Document doc = loadFixture("dc_post.html");
|
||||
|
||||
PostDetail detail = new DcPostDetailParser().parse(doc);
|
||||
|
||||
assertThat(Jsoup.parse(detail.content()).text()).contains("출퇴근 소요 시간");
|
||||
assertThat(detail.content()).contains("<img");
|
||||
assertThat(detail.imageUrls()).isNotEmpty();
|
||||
assertThat(detail.imageUrls().get(0)).contains("dcinside.co.kr");
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user