基于 Jsoup的海投網(wǎng)爬蟲

package com.zzz.jsouplib;

import org.jsoup.Jsoup;
import org.jsoup.nodes.Document;
import org.jsoup.nodes.Element;
import org.jsoup.select.Elements;
import java.io.IOException;
import java.util.ArrayList;
import java.util.List;
import java.util.function.Consumer;

public class JsoupTest {
    public static void main(String[] args) {
        List<CompanyJobModel> companyJobModelList = new ArrayList<>();
        //一次性扒20頁的數(shù)據(jù)
        for (int pageNum = 1; pageNum <= 20; pageNum++) {
            try {
                Document doc = Jsoup.connect("https://xyzp.haitou.cc/page-" + pageNum)
                        .data("query", "Java")
                        .userAgent("Mozilla")
                        .get();
                Element bodyElement = doc.getElementsByTag("body").get(0);
                Element scopeElement = bodyElement.getElementsByTag("div").get(0);
                Element pageWrapper = scopeElement.getElementById("page-wrapper");
                Element pageMain = pageWrapper.getElementsByClass("page-main").get(0);
                Element pagePanel = pageMain.getElementsByClass("panel panel-default").get(0);
                Element pagePanel2 = pagePanel.getElementsByClass("panel-body remove padding top bottom").get(0);
                Element gridView = pagePanel2.getElementById("w0");
                Element table = gridView.getElementsByClass("table cxxt-table").get(0);
                Element tbody = table.getElementsByTag("tbody").get(0);
                Elements datas = tbody.getElementsByTag("tr");

                for (Element element : datas
                ) {
                    //該公司的data-key
                    String dataKey = element.attr("data-key");
                    String title = element
                            .getElementsByClass("cxxt-title").get(0)
                            .getElementsByTag("a").get(0)
                            .getElementsByTag("div").get(0)
                            .text();
                    //發(fā)布時間
                    String time = element.getElementsByClass("cxxt-time").text();
                    String detailUrl = "https://xyzp.haitou.cc/article/" + dataKey + ".html";

                    //職位列表
                    final List<String> careerList = new ArrayList<>();

                    //職位的a標(biāo)簽的列表
                    Elements careerTagList = element.getElementsByClass("text-ellipsis cxxt-position")
                            .get(0)
                            .getElementsByTag("a");

                    //遍歷職位的a標(biāo)簽的列表
                    careerTagList.forEach(new Consumer<Element>() {
                        @Override
                        public void accept(Element element) {
                            String career = element.getElementsByTag("span").get(0).text();
                            careerList.add(career);
                        }
                    });

                    //涉及城市列表
                    final List<String> cityList = new ArrayList<>();
                    //涉及城市的span標(biāo)簽的列表
                    Element citySpanTag = element.getElementsByClass("text-ellipsis")
                            .get(1)
                            .getElementsByTag("span")
                            .get(0);
                    if (citySpanTag.text().equals("不限")) {//當(dāng)城市的span tag值為不限
                        cityList.add("不限");
                    } else {
                        //城市的span tag 沒有值 繼續(xù)遍歷它的子tag
                        //涉及城市的a標(biāo)簽的列表
                        Elements citySpanElements = citySpanTag.getElementsByTag("span");
                        for (int i = 1; i <= citySpanElements.size() - 1; i++) {
                            //排除掉第一項(xiàng) 只要子tag
                            String city = citySpanElements.get(i).text();
                            cityList.add(city);
                        }
                    }

                    //將扒取的信息構(gòu)建成model
                    CompanyJobModel companyJobModel = new CompanyJobModel(
                            dataKey,
                            title,
                            time,
                            detailUrl,
                            careerList,
                            cityList
                    );
                    //將model添加到扒取列表
                    companyJobModelList.add(companyJobModel);
                }
            } catch (IOException e) {
                e.printStackTrace();
            }
        }
        //查看扒到的所有信息
        System.out.print("list" + companyJobModelList);
    }
}



public class CompanyJobModel {
    private String dataKey;
    private String title;
    private String time;
    private String detailUrl;
    private List<String> careerList;
    private List<String> cityList;

    public CompanyJobModel(
            String dataKey,
            String title,
            String time,
            String detailUrl,
            List<String> careerList,
            List<String> cityList
    ) {
        this.dataKey = dataKey;
        this.title = title;
        this.time = time;
        this.detailUrl = detailUrl;
        this.careerList = careerList;
        this.cityList = cityList;
    }

    @Override
    public String toString() {
        return "dataKey: " + dataKey + "\n" +
                "title: " + title + "\n" +
                "time: " + time + "\n" +
                "detailUrl: " + detailUrl + "\n" +
                "careerList: " + careerList + "\n" +
                "cityList: " + cityList + "\n";
    }
}

項(xiàng)目地址 https://github.com/ZYF99/HaitouApp

最后編輯于
?著作權(quán)歸作者所有,轉(zhuǎn)載或內(nèi)容合作請聯(lián)系作者
【社區(qū)內(nèi)容提示】社區(qū)部分內(nèi)容疑似由AI輔助生成,瀏覽時請結(jié)合常識與多方信息審慎甄別。
平臺聲明:文章內(nèi)容(如有圖片或視頻亦包括在內(nèi))由作者上傳并發(fā)布,文章內(nèi)容僅代表作者本人觀點(diǎn),簡書系信息發(fā)布平臺,僅提供信息存儲服務(wù)。

友情鏈接更多精彩內(nèi)容