Web Scraping with rvest and R

妖精的绣舞 提交于 2019-12-25 08:13:15

问题


I am trying to web scrape the total assets of a particular fund in this case ADAFX from http://www.morningstar.com/funds/xnas/adafx/quote.html. But the result is always charecter (empty); what am I doing wrong?

I have used rvest before with mixed results, so I figured time to get expert help from the community of trusted gurus (thats you).

library(rvest)      
Symbol.i ="ADAFX"
url <-Paste("http://www.morningstar.com/funds/xnas/",Symbol.i,"/quote.html",sep="")
  tryCatch(NetAssets.i <- url %>%
             read_html() %>%
             html_nodes(xpath='//*[@id="gr_total_asset_wrap"]/span/span') %>%
             html_text(), error = function(e) NetAssets.i = NA)

Thank you in advance, Cheers,

Aaron Soderstrom


回答1:


It's a dynamic page that loads data for the various sectinons via XHR requests, so you have to look at the Developer Tools Network tab to get the target content URLs.

library(httr)
library(rvest)

res <- GET(url = "http://quotes.morningstar.com/fundq/c-header",
           query = list(
             t="XNAS:ADAFX",
             region="usa",
             culture="en-US",
             version="RET",
             test="QuoteiFrame"
           )
)

content(res) %>%
  html_nodes("span[vkey='TotalAssets']") %>%
  html_text() %>%
  trimws()
## [1] "20.6  mil"



回答2:


Here is the csv file it calls.

library(httr)
library(rvest)
library(tm)
library(plyr)
require("dplyr")

MF.List <- read.csv("C:/Users/Aaron/Documents/Investment Committee/Screener/Filtered Funds.csv")
Category.list <- read.csv("C:/Users/Aaron/Documents/Investment Committee/Screener/Category.csv")
Category.list <- na.omit(Category.list)

Category.name <- "Financial"
MF.Category.List <- filter(MF.List, Category == Category.name)

morningstar.scrape <- list()

for(i in 1:nrow(MF.Category.List)){

  Symbol.i =as.character(MF.Category.List[i,"Symbol"])
  res <- GET(url = "http://quotes.morningstar.com/fundq/c-header",
             query = list(
               t=paste("XNAS:",Symbol.i,sep=""),
               region="usa",
               culture="en-US",
               version="RET",
               test="QuoteiFrame"
             )
  )

  tryCatch(
    TTM.Yield <- content(res) %>%
      html_nodes("span[vkey='ttmYield']") %>%
      html_text() %>%
      trimws()
    , error = function(e) TTM.Yield<-NA)

  tryCatch(
    Load <- content(res) %>%
      html_nodes("span[vkey='Load']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Load = NA)

  tryCatch(
    Total.Assets <- content(res) %>%
      html_nodes("span[vkey='TotalAssets']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Total.Assets = NA)

  tryCatch(
    Expense.Ratio <- content(res) %>%
      html_nodes("span[vkey='ExpenseRatio']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Expense.Ratio = NA)

  tryCatch(
    Fee.Level <- content(res) %>%
      html_nodes("span[vkey='FeeLevel']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Fee.Level = NA)

  tryCatch(
    Turnover <- content(res) %>%
      html_nodes("span[vkey='Turnover']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Turnover = NA)

  tryCatch(
    Status <- content(res) %>%
      html_nodes("span[vkey='Status']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Status = NA)

  tryCatch(
    Min.Investment <- content(res) %>%
      html_nodes("span[vkey='MinInvestment']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Min.Investment = NA)

  tryCatch(
    Yield.30day <- content(res) %>%
      html_nodes("span[vkey='Yield']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Yield.30day = NA)

  tryCatch(
    Investment.Style <- content(res) %>%
      html_nodes("span[vkey='InvestmentStyle']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Investment.Style = NA)

  tryCatch(
    Bond.Style <- content(res) %>%
      html_nodes("span[vkey='BondStyle']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Bond.Style = NA)

  x.frame <- c(Symbol =as.character(Symbol.i),TTM.Yield = as.character(TTM.Yield), Load = as.character(Load),
               Total.Assets = as.character(Total.Assets),Expense.Ratio = as.character(Expense.Ratio),
               Turnover = as.character(Turnover), Status = as.character(Status), 
               Yield.30day = as.character(Yield.30day), 
               Investment.Style = as.character(Investment.Style),Bond.Style = as.character(Bond.Style))

  morningstar.scrape[[i]] = x.frame
  x.frame = NULL
}

MS.scrape <- do.call(rbind, morningstar.scrape)



回答3:


Working code,

I added a function to the web scrape and removed the library(tm).

library(httr)
library(rvest)


    get.morningstar <- function(Symbol.i,htmlnode){
      res <- GET(url = "http://quotes.morningstar.com/fundq/c-header",
                 query = list(
                   t=paste("XNAS:",Symbol.i,sep=""),
                   region="usa",
                   culture="en-US",
                   version="RET",
                   test="QuoteiFrame"
                 )
      )

      x <- content(res) %>%
        html_nodes(htmlnode) %>%
        html_text() %>%
        trimws()

      return(x)
    }



    MF.List <- read.csv("C:/Users/Aaron/Documents/Bitrix24/Investment Committee/Screener/Filtered Funds.csv")
    Category.list <- read.csv("C:/Users/Aaron/Documents/Bitrix24/Investment Committee/Screener/Category.csv")
    Category.list <- na.omit(Category.list)

    Category.name <- "Small Growth"
    MF.Category.List <- MF.List[grepl(Category.name,MF.List$Category), ]
    morningstar.scrape <- list()

    for(i in 1:nrow(MF.Category.List)){
      Symbol.i =as.character(MF.Category.List[i,"Symbol"])
      try(Total.Assets <- get.morningstar(Symbol.i,"span[vkey='TotalAssets']"))
      print(Total.Assets)
    }


来源:https://stackoverflow.com/questions/42359680/web-scraping-with-rvest-and-r

易学教程内所有资源均来自网络或用户发布的内容,如有违反法律规定的内容欢迎反馈
该文章没有解决你所遇到的问题?点击提问,说说你的问题,让更多的人一起探讨吧!