mirror of
https://github.com/headporter81/specialsource-homepage-backend.git
synced 2026-10-07 15:21:09 +09:00
45 lines
1.6 KiB
Java
45 lines
1.6 KiB
Java
package company.specialsource.scraper;
|
|
|
|
import org.jsoup.Jsoup;
|
|
import org.jsoup.nodes.Document;
|
|
import org.junit.jupiter.api.Test;
|
|
|
|
import java.io.IOException;
|
|
import java.io.InputStream;
|
|
|
|
import static org.assertj.core.api.Assertions.assertThat;
|
|
|
|
class DetailParserTest {
|
|
|
|
private Document loadFixture(String fileName) throws IOException {
|
|
try (InputStream in = getClass().getResourceAsStream("/scraper/" + fileName)) {
|
|
assertThat(in).as("fixture %s must exist", fileName).isNotNull();
|
|
return Jsoup.parse(in, "UTF-8", "https://bbs.ruliweb.com/");
|
|
}
|
|
}
|
|
|
|
@Test
|
|
void ruliwebPostDetailParser_extractsTextAndImages() throws IOException {
|
|
Document doc = loadFixture("ruliweb_post_market.html");
|
|
|
|
PostDetail detail = new RuliwebPostDetailParser().parse(doc);
|
|
|
|
assertThat(Jsoup.parse(detail.content()).text()).contains("핫딜관리자입니다");
|
|
assertThat(detail.content()).contains("<img");
|
|
assertThat(detail.imageUrls()).isNotEmpty();
|
|
assertThat(detail.imageUrls().get(0)).startsWith("https://");
|
|
}
|
|
|
|
@Test
|
|
void dcPostDetailParser_extractsTextAndImages() throws IOException {
|
|
Document doc = loadFixture("dc_post.html");
|
|
|
|
PostDetail detail = new DcPostDetailParser().parse(doc);
|
|
|
|
assertThat(Jsoup.parse(detail.content()).text()).contains("출퇴근 소요 시간");
|
|
assertThat(detail.content()).contains("<img");
|
|
assertThat(detail.imageUrls()).isNotEmpty();
|
|
assertThat(detail.imageUrls().get(0)).contains("dcinside.co.kr");
|
|
}
|
|
}
|