数据挖掘第五次

2024-02-24 11:30:52

作业一

作业要求：

熟练掌握 Selenium 查找HTML元素、爬取Ajax网页数据、等待HTML元素等内容。
使用Selenium框架爬取京东商城某类商品信息及图片。
候选网站：http://www.jd.com/

实验过程：

驱动配置

chrome_options = Options()
chrome_options.add_argument("——headless")
chrome_options.add_argument("——disable-gpu")
self.driver = webdriver.Chrome(chrome_options=chrome_options)

数据获得

lis = self.driver.find_elements_by_xpath("//div[@id='J_goodsList']//li[@class='gl-item']")
time.sleep(1)
for li in lis:
    time.sleep(1)
    try:
        src1 = li.find_element_by_xpath(".//div[@class='p-img']//a//img").get_attribute("src")
        time.sleep(1)
    except:
        src1 = ""
    try:
        src2 = li.find_element_by_xpath(".//div[@class='p-img']//a//img").get_attribute("data-lazy-img")
        time.sleep(1)
    except:
        src2 = ""
    try:
        price = li.find_element_by_xpath(".//div[@class='p-price']//i").text
        time.sleep(1)
    except:
        price = "0"

    note = li.find_element_by_xpath(".//div[@class='p-name p-name-type-2']//em").text
    mark = note.split(" ")[0]
    mark = mark.replace("爱心东东\n", "")
    mark = mark.replace(",", "")
    note = note.replace("爱心东东\n", "")
    note = note.replace(",", "")

保存图片

if src1:
    src1 = urllib.request.urljoin(self.driver.current_url, src1)
    p = src1.rfind(".")
    mFile = no + src1[p:]
elif src2:
    src2 = urllib.request.urljoin(self.driver.current_url, src2)
    p = src2.rfind(".")
    mFile = no + src2[p:]
if src1 or src2:
    T = threading.Thread(target=self.downloadDB, args=(src1, src2, mFile))
    T.setDaemon(False)
    T.start()
    self.threads.append(T)
else:
    mFile = ""

数据插入

def insertDB(self, mNo, mMark, mPrice, mNote, mFile):
    try:
        sql = "insert into phones (mNo,mMark,mPrice,mNote,mFile) values (?,?,?,?,?)"

        self.cursor.execute(sql, (mNo, mMark, mPrice, mNote, mFile))
    except Exception as err:
        print(err)

图片下载

def downloadDB(self, src1, src2, mFile):
    data = None
    if src1:
        try:
            req = urllib.request.Request(src1, headers=JD.header)
            resp = urllib.request.urlopen(req, timeout=100)
            data = resp.read()
        except:
            pass
    if not data and src2:
        try:
            req = urllib.request.Request(src2, headers=JD.header)
            resp = urllib.request.urlopen(req, timeout=100)
            data = resp.read()
        except:
            pass
    if data:
        print("download begin!", mFile)
        fobj = open(JD.imagepath + "\\" + mFile, "wb")
        fobj.write(data)
        fobj.close()
        print("download finish!", mFile)

实验结果：

实验心得：这次实验是课堂代码的复现，通过这次实验我加深了对Selenium的理解，已经xpath的掌握

码云地址：作业5/main.py · 刘洋/2019数据采集与融合 - 码云 - 开源中国 (gitee.com)

作业二

作业要求：

熟练掌握 Selenium 查找HTML元素、实现用户模拟登录、爬取Ajax网页数据、等待HTML元素等内容。
使用Selenium框架+MySQL爬取中国mooc网课程资源信息（课程号、课程名称、教学
进度、课程状态，课程图片地址），同时存储图片到本地项目根目录下的imgs文件夹
中，图片的名称用课程名来存储。
候选网站：中国mooc网：https://www.icourse163.org

实验过程：

发送请求

option = webdriver.ChromeOptions()
option.add_experimental_option("detach", True)
driver = webdriver.Chrome(chrome_options=option)
#请求
driver.get('https://www.icourse163.org/')

数据库部分，包括插入数据的实现

class CSDB:
    con = pymysql.connect(host="127.0.0.1", port=3306, user="root", passwd="20010109l?y!", db="class",
                          charset="utf8")
    cursor = con.cursor(pymysql.cursors.DictCursor)

    def createDB(self):
        try:
            self.cursor.execute('create table Course(Id varchar (10),cCourse varchar (40),cCollege varchar (20),cSchedule varchar (30),'
                            'cCourseStatus varchar (30), clmgUrl varchar (255))')
        except Exception as err:
            print(err)
            print(1)

    def insert(self,id,course,college,schedule,coursestatus,clmgurl):
        try:
            self.cursor.execute('insert into Course(Id,cCourse,cCollege,cSchedule,cCourseStatus,clmgUrl) '
                            'values (%s,%s,%s,%s,%s,%s)', (id, course, college, schedule, coursestatus, clmgurl))
        except Exception as err:
            print(err)

    def closeDB(self):
        self.con.commit()
        self.con.close()

保存照片，照片名为课程名，照片保存在images文件夹中

def download(url,name):
    req = urllib.request.Request(url,)
    data = urllib.request.urlopen(req, timeout=100)
    data = data.read()
    fobj = open(r"images/" + str(name) + ".jpg", "wb")
    fobj.write(data)
    fobj.close()
    print("downloaded" + (name) + ".jpg")

通过点击操作进入登录页面，在登录页面进行扫码登录，然后通过点击进入我的课程页面，获取数据

sign = driver.find_element(By.XPATH, '/html/body/div[4]/div[2]/div[1]/div/div/div[1]/div[3]/div[3]/div').click()
time.sleep(10)
#进入我的课堂
find = driver.find_element(By.XPATH, '//*[@id="j-indexNav-bar"]/div/div/div/div/div[7]/div[3]/div/div/a/span')
# 图标被遮挡，导致无法直接用find.click()点击
driver.execute_script("arguments[0].click();", find)

数据的获取

#课程名
course = driver.find_elements(By.XPATH, '//*[@id="j-coursewrap"]/div/div[1]/div/div[1]/a/div[2]/div[1]/div[1]/div/span[2]')
#学校
college = driver.find_elements(By.XPATH, '//*[@id="j-coursewrap"]/div/div[1]/div/div[1]/a/div[2]/div[1]/div[2]/a')
#课时
sche = driver.find_elements(By.XPATH, '//*[@id="j-coursewrap"]/div/div[1]/div/div[1]/a/div[2]/div[2]/div[1]/div[1]/div[1]/a/span')
#状态
status = driver.find_elements(By.XPATH, '/html/body/div[4]/div[2]/div[3]/div/div[1]/div[3]/div/div[2]/div/div/div[2]/div[1]/div[2]/div/div[1]/div/div[1]/a/div[2]/div[2]/div[2]')
#url
url = driver.find_elements(By.XPATH, '/html/body/div[4]/div[2]/div[3]/div/div[1]/div[3]/div/div[2]/div/div/div[2]/div[1]/div[2]/div/div[1]/div/div[1]/a/div[1]/img')

实验结果：