Web Scraping with rvest and R

2019-09-06 17:36发布

I am trying to web scrape the total assets of a particular fund in this case ADAFX from http://www.morningstar.com/funds/xnas/adafx/quote.html. But the result is always charecter (empty); what am I doing wrong?

I have used rvest before with mixed results, so I figured time to get expert help from the community of trusted gurus (thats you).

library(rvest)      
Symbol.i ="ADAFX"
url <-Paste("http://www.morningstar.com/funds/xnas/",Symbol.i,"/quote.html",sep="")
  tryCatch(NetAssets.i <- url %>%
             read_html() %>%
             html_nodes(xpath='//*[@id="gr_total_asset_wrap"]/span/span') %>%
             html_text(), error = function(e) NetAssets.i = NA)

Thank you in advance, Cheers,

Aaron Soderstrom

3条回答
何必那么认真
2楼-- · 2019-09-06 18:20

Working code,

I added a function to the web scrape and removed the library(tm).

library(httr)
library(rvest)


    get.morningstar <- function(Symbol.i,htmlnode){
      res <- GET(url = "http://quotes.morningstar.com/fundq/c-header",
                 query = list(
                   t=paste("XNAS:",Symbol.i,sep=""),
                   region="usa",
                   culture="en-US",
                   version="RET",
                   test="QuoteiFrame"
                 )
      )

      x <- content(res) %>%
        html_nodes(htmlnode) %>%
        html_text() %>%
        trimws()

      return(x)
    }



    MF.List <- read.csv("C:/Users/Aaron/Documents/Bitrix24/Investment Committee/Screener/Filtered Funds.csv")
    Category.list <- read.csv("C:/Users/Aaron/Documents/Bitrix24/Investment Committee/Screener/Category.csv")
    Category.list <- na.omit(Category.list)

    Category.name <- "Small Growth"
    MF.Category.List <- MF.List[grepl(Category.name,MF.List$Category), ]
    morningstar.scrape <- list()

    for(i in 1:nrow(MF.Category.List)){
      Symbol.i =as.character(MF.Category.List[i,"Symbol"])
      try(Total.Assets <- get.morningstar(Symbol.i,"span[vkey='TotalAssets']"))
      print(Total.Assets)
    }
查看更多
看我几分像从前
3楼-- · 2019-09-06 18:27

It's a dynamic page that loads data for the various sectinons via XHR requests, so you have to look at the Developer Tools Network tab to get the target content URLs.

library(httr)
library(rvest)

res <- GET(url = "http://quotes.morningstar.com/fundq/c-header",
           query = list(
             t="XNAS:ADAFX",
             region="usa",
             culture="en-US",
             version="RET",
             test="QuoteiFrame"
           )
)

content(res) %>%
  html_nodes("span[vkey='TotalAssets']") %>%
  html_text() %>%
  trimws()
## [1] "20.6  mil"
查看更多
劫难
4楼-- · 2019-09-06 18:30

Here is the csv file it calls.

library(httr)
library(rvest)
library(tm)
library(plyr)
require("dplyr")

MF.List <- read.csv("C:/Users/Aaron/Documents/Investment Committee/Screener/Filtered Funds.csv")
Category.list <- read.csv("C:/Users/Aaron/Documents/Investment Committee/Screener/Category.csv")
Category.list <- na.omit(Category.list)

Category.name <- "Financial"
MF.Category.List <- filter(MF.List, Category == Category.name)

morningstar.scrape <- list()

for(i in 1:nrow(MF.Category.List)){

  Symbol.i =as.character(MF.Category.List[i,"Symbol"])
  res <- GET(url = "http://quotes.morningstar.com/fundq/c-header",
             query = list(
               t=paste("XNAS:",Symbol.i,sep=""),
               region="usa",
               culture="en-US",
               version="RET",
               test="QuoteiFrame"
             )
  )

  tryCatch(
    TTM.Yield <- content(res) %>%
      html_nodes("span[vkey='ttmYield']") %>%
      html_text() %>%
      trimws()
    , error = function(e) TTM.Yield<-NA)

  tryCatch(
    Load <- content(res) %>%
      html_nodes("span[vkey='Load']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Load = NA)

  tryCatch(
    Total.Assets <- content(res) %>%
      html_nodes("span[vkey='TotalAssets']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Total.Assets = NA)

  tryCatch(
    Expense.Ratio <- content(res) %>%
      html_nodes("span[vkey='ExpenseRatio']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Expense.Ratio = NA)

  tryCatch(
    Fee.Level <- content(res) %>%
      html_nodes("span[vkey='FeeLevel']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Fee.Level = NA)

  tryCatch(
    Turnover <- content(res) %>%
      html_nodes("span[vkey='Turnover']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Turnover = NA)

  tryCatch(
    Status <- content(res) %>%
      html_nodes("span[vkey='Status']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Status = NA)

  tryCatch(
    Min.Investment <- content(res) %>%
      html_nodes("span[vkey='MinInvestment']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Min.Investment = NA)

  tryCatch(
    Yield.30day <- content(res) %>%
      html_nodes("span[vkey='Yield']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Yield.30day = NA)

  tryCatch(
    Investment.Style <- content(res) %>%
      html_nodes("span[vkey='InvestmentStyle']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Investment.Style = NA)

  tryCatch(
    Bond.Style <- content(res) %>%
      html_nodes("span[vkey='BondStyle']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Bond.Style = NA)

  x.frame <- c(Symbol =as.character(Symbol.i),TTM.Yield = as.character(TTM.Yield), Load = as.character(Load),
               Total.Assets = as.character(Total.Assets),Expense.Ratio = as.character(Expense.Ratio),
               Turnover = as.character(Turnover), Status = as.character(Status), 
               Yield.30day = as.character(Yield.30day), 
               Investment.Style = as.character(Investment.Style),Bond.Style = as.character(Bond.Style))

  morningstar.scrape[[i]] = x.frame
  x.frame = NULL
}

MS.scrape <- do.call(rbind, morningstar.scrape)
查看更多
登录 后发表回答