【问题标题】:findElement with XPath works line by line but it fails in a loop使用 XPath 的 findElement 逐行工作,但在循环中失败
【发布时间】:2021-03-16 17:18:14
【问题描述】:

我想从这个 website 收集电子邮件我创建了这个循环,当我单独运行每个部分时它可以工作,但当它一起运行时它不起作用。

library(RSelenium)


#######################################University College Dublin
dep<-"https://people.ucd.ie/search?by=text"
rD <- rsDriver(browser="firefox", port=4545L, verbose=F)
remDr <- rD[["client"]]
remDr$navigate(dep)

mail<-list()

for(i in 2:65){
  if(i==2){
    webElem <- remDr$findElement(using = 'xpath', '//*[@id="app"]/div/div/main/div[2]/div[3]/div[2]/div[1]/div[3]/span[2]')  
    webElem$clickElement()
    
  }else{
  w<-  paste0('//*[@id="app"]/div/div/main/div[2]/div[3]/div[2]/div[1]/div[3]/span[', i,"]/button" )
  webElem <- remDr$findElement(using = 'xpath', w)  
  webElem$clickElement()
  }
  
  for(j in 1:25){
    #click each person 25 x page
    ww<-paste0('//*[@id="app"]/div/div/main/div[2]/div[3]/div[2]/div[4]/div[', j,"]/div[1]/div[2]/div[1]/a" )
    webElem <- remDr$findElement(using = 'xpath', ww)  
    webElem$clickElement()
    #click emails
    webElem <- remDr$findElement(using = 'xpath', '//*[@id="app"]/div/div/main/div[1]/div[3]/div[3]/div[2]/div[1]/div[2]/span/a')  
    ma<-webElem$getElementText()
    if(length(ma)!=0){mail<-c(mail,ma)}
    webElem$goBack() 
    rm(ma)
  }
 }

【问题讨论】:

  • 谢谢@Konrad Rudolph 现在它更加清晰和健全。你知道我为什么得到这个吗?在这种情况下是否有不同的 xpath 可以调用?你有代码建议吗?

标签: r web-scraping rvest rselenium


【解决方案1】:

这里有一个可能的解决方案,但是非常耗时,有更好的选择吗?

dep<-"https://people.ucd.ie/search?by=text"
rD <- rsDriver(browser="firefox", port=4545L, verbose=F)
remDr <- rD[["client"]]

mail<-list()
remDr$navigate(dep)
peo<-paste0('.userStub__userStub___ju2wK:nth-child(', 1:25, ') a')

for(i in 1:63){
for(j in 1:25){
  #.userStub__userStub___ju2wK:nth-child(1) a
  #.userStub__userStub___ju2wK:nth-child(2) a
  #.userStub__userStub___ju2wK:nth-child(25) a
  webElem <- remDr$findElement(using = 'css selector', peo[j])  
  webElem$clickElement()
  Sys.sleep(1) #time to load the page
  
  r<-webElem$getPageSource() #get all webpage text and selct mailto
  r<-unlist(str_split(as.character(r),'"'))
  w<-which(grepl("mailto:", r))
  
  if(length(w)!=0){
  a<-r[w]
  a<-gsub("mailto:", "", a, fixed = T)
  mail<-c(mail, a)
  }
  
  #go back
  webElem <- remDr$findElement(using = 'css selector', '#app > div > div > main > div.hero__hero___3_ZZJ > a')  
  webElem$clickElement()
  Sys.sleep(1) 
 # #app > div > div > main > div.hero__hero___3_ZZJ > a > div > span
  ##app > div > div > main > div.hero__hero___3_ZZJ > a
}
  
#next page
  webElem <- remDr$findElement(using = 'css selector', '#app > div > div > main > div.results__resultsContainer___18wNx > div.results__paginatedUsersContainer___3PU1S > div:nth-child(3) > div:nth-child(1) > div.paginationBar__paginationItems___3UjZC > span:nth-child(6)')  
  webElem$clickElement()
  Sys.sleep(1) 
  
}

【讨论】:

    猜你喜欢
    • 1970-01-01
    • 1970-01-01
    • 2020-08-05
    • 2020-08-04
    • 2019-04-09
    • 1970-01-01
    • 1970-01-01
    • 1970-01-01
    • 1970-01-01
    相关资源
    最近更新 更多