Web Scraping with rvest and R

I am trying to web scrape the total assets of a particular fund in this case ADAFX from http://www.morningstar.com/funds/xnas/adafx/quote.html. But the result is always charecter (empty); what am I doing wrong?

I have used rvest before with mixed results, so I figured time to get expert help from the community of trusted gurus (thats you).

library(rvest)      
Symbol.i ="ADAFX"
url <-Paste("http://www.morningstar.com/funds/xnas/",Symbol.i,"/quote.html",sep="")
  tryCatch(NetAssets.i <- url %>%
             read_html() %>%
             html_nodes(xpath='//*[@id="gr_total_asset_wrap"]/span/span') %>%
             html_text(), error = function(e) NetAssets.i = NA)

Thank you in advance, Cheers,

Aaron Soderstrom

标签： r web-scraping rvest

3条回答

何必那么认真

2楼-- · 2019-09-06 18:20

Working code,

I added a function to the web scrape and removed the library(tm).

library(httr)
library(rvest)


    get.morningstar <- function(Symbol.i,htmlnode){
      res <- GET(url = "http://quotes.morningstar.com/fundq/c-header",
                 query = list(
                   t=paste("XNAS:",Symbol.i,sep=""),
                   region="usa",
                   culture="en-US",
                   version="RET",
                   test="QuoteiFrame"
                 )
      )

      x <- content(res) %>%
        html_nodes(htmlnode) %>%
        html_text() %>%
        trimws()

      return(x)
    }



    MF.List <- read.csv("C:/Users/Aaron/Documents/Bitrix24/Investment Committee/Screener/Filtered Funds.csv")
    Category.list <- read.csv("C:/Users/Aaron/Documents/Bitrix24/Investment Committee/Screener/Category.csv")
    Category.list <- na.omit(Category.list)

    Category.name <- "Small Growth"
    MF.Category.List <- MF.List[grepl(Category.name,MF.List$Category), ]
    morningstar.scrape <- list()

    for(i in 1:nrow(MF.Category.List)){
      Symbol.i =as.character(MF.Category.List[i,"Symbol"])
      try(Total.Assets <- get.morningstar(Symbol.i,"span[vkey='TotalAssets']"))
      print(Total.Assets)
    }

0人赞添加讨论(0) 举报

看我几分像从前

3楼-- · 2019-09-06 18:27

It's a dynamic page that loads data for the various sectinons via XHR requests, so you have to look at the Developer Tools Network tab to get the target content URLs.

library(httr)
library(rvest)

res <- GET(url = "http://quotes.morningstar.com/fundq/c-header",
           query = list(
             t="XNAS:ADAFX",
             region="usa",
             culture="en-US",
             version="RET",
             test="QuoteiFrame"
           )
)

content(res) %>%
  html_nodes("span[vkey='TotalAssets']") %>%
  html_text() %>%
  trimws()
## [1] "20.6  mil"

0人赞添加讨论(0) 举报

劫难

4楼-- · 2019-09-06 18:30

Here is the csv file it calls.

library(httr)
library(rvest)
library(tm)
library(plyr)
require("dplyr")

MF.List <- read.csv("C:/Users/Aaron/Documents/Investment Committee/Screener/Filtered Funds.csv")
Category.list <- read.csv("C:/Users/Aaron/Documents/Investment Committee/Screener/Category.csv")
Category.list <- na.omit(Category.list)

Category.name <- "Financial"
MF.Category.List <- filter(MF.List, Category == Category.name)

morningstar.scrape <- list()

for(i in 1:nrow(MF.Category.List)){

  Symbol.i =as.character(MF.Category.List[i,"Symbol"])
  res <- GET(url = "http://quotes.morningstar.com/fundq/c-header",
             query = list(
               t=paste("XNAS:",Symbol.i,sep=""),
               region="usa",
               culture="en-US",
               version="RET",
               test="QuoteiFrame"
             )
  )

  tryCatch(
    TTM.Yield <- content(res) %>%
      html_nodes("span[vkey='ttmYield']") %>%
      html_text() %>%
      trimws()
    , error = function(e) TTM.Yield<-NA)

  tryCatch(
    Load <- content(res) %>%
      html_nodes("span[vkey='Load']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Load = NA)

  tryCatch(
    Total.Assets <- content(res) %>%
      html_nodes("span[vkey='TotalAssets']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Total.Assets = NA)

  tryCatch(
    Expense.Ratio <- content(res) %>%
      html_nodes("span[vkey='ExpenseRatio']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Expense.Ratio = NA)

  tryCatch(
    Fee.Level <- content(res) %>%
      html_nodes("span[vkey='FeeLevel']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Fee.Level = NA)

  tryCatch(
    Turnover <- content(res) %>%
      html_nodes("span[vkey='Turnover']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Turnover = NA)

  tryCatch(
    Status <- content(res) %>%
      html_nodes("span[vkey='Status']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Status = NA)

  tryCatch(
    Min.Investment <- content(res) %>%
      html_nodes("span[vkey='MinInvestment']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Min.Investment = NA)

  tryCatch(
    Yield.30day <- content(res) %>%
      html_nodes("span[vkey='Yield']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Yield.30day = NA)

  tryCatch(
    Investment.Style <- content(res) %>%
      html_nodes("span[vkey='InvestmentStyle']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Investment.Style = NA)

  tryCatch(
    Bond.Style <- content(res) %>%
      html_nodes("span[vkey='BondStyle']") %>%
      html_text() %>%
      trimws()
    , error = function(e) Bond.Style = NA)

  x.frame <- c(Symbol =as.character(Symbol.i),TTM.Yield = as.character(TTM.Yield), Load = as.character(Load),
               Total.Assets = as.character(Total.Assets),Expense.Ratio = as.character(Expense.Ratio),
               Turnover = as.character(Turnover), Status = as.character(Status), 
               Yield.30day = as.character(Yield.30day), 
               Investment.Style = as.character(Investment.Style),Bond.Style = as.character(Bond.Style))

  morningstar.scrape[[i]] = x.frame
  x.frame = NULL
}

MS.scrape <- do.call(rbind, morningstar.scrape)

0人赞添加讨论(0) 举报

Web Scraping with rvest and R

采纳回答

编辑标签

举报内容

检举类型

检举原因

检举说明(必填)

打开微信“扫一扫”，打开网页后点击屏幕右上角分享按钮

付费偷看金额在0.1-10元之间