Compare commits
2
Commits
feed583fa9
...
502dc45781
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
502dc45781 | ||
|
|
4b3e85be62 |
+1
-1
@@ -4,7 +4,7 @@
|
|||||||
/inst/reports/
|
/inst/reports/
|
||||||
!/inst/reports/*.pdf
|
!/inst/reports/*.pdf
|
||||||
!/inst/reports/*.tex
|
!/inst/reports/*.tex
|
||||||
/data/csv/*
|
/inst/csv/*
|
||||||
/parlament_49_53_texts/
|
/parlament_49_53_texts/
|
||||||
.Rproj.user
|
.Rproj.user
|
||||||
*.Rproj
|
*.Rproj
|
||||||
|
|||||||
@@ -41,7 +41,7 @@ fetch_batch <- function(offset, download_dir) {
|
|||||||
#' if create is TRUE, the directory given in download_dir is created
|
#' if create is TRUE, the directory given in download_dir is created
|
||||||
#'
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
fetch_all <- function(download_dir="data/records/", create=FALSE) {
|
fetch_all <- function(download_dir="inst/records/", create=FALSE) {
|
||||||
# check if download_dir path is a directory path
|
# check if download_dir path is a directory path
|
||||||
if (str_sub(download_dir, -1) != .Platform$file.sep)
|
if (str_sub(download_dir, -1) != .Platform$file.sep)
|
||||||
download_dir <- str_c(download_dir, .Platform$file.sep)
|
download_dir <- str_c(download_dir, .Platform$file.sep)
|
||||||
|
|||||||
@@ -8,7 +8,7 @@
|
|||||||
#' @param path character
|
#' @param path character
|
||||||
#'
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
read_all <- function(path="data/records/") {
|
read_all <- function(path="inst/records/") {
|
||||||
cat("Reading all records from", path, "\n")
|
cat("Reading all records from", path, "\n")
|
||||||
available_protocols <- list.files(path)
|
available_protocols <- list.files(path)
|
||||||
res <- pblapply(available_protocols, read_one, path=path)
|
res <- pblapply(available_protocols, read_one, path=path)
|
||||||
@@ -212,7 +212,7 @@ parse_speakerlist <- function(speakerliste_xml) {
|
|||||||
#' @param create set TRUE if the path does not exist yet and you want to create it
|
#' @param create set TRUE if the path does not exist yet and you want to create it
|
||||||
#'
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
write_to_csv <- function(tables, path="data/csv/", create=F) {
|
write_to_csv <- function(tables, path="inst/csv/", create=F) {
|
||||||
check_directory(path, create)
|
check_directory(path, create)
|
||||||
write.table(tables$speaker, str_c(path, "speaker.csv"))
|
write.table(tables$speaker, str_c(path, "speaker.csv"))
|
||||||
write.table(tables$speeches, str_c(path, "speeches.csv"))
|
write.table(tables$speeches, str_c(path, "speeches.csv"))
|
||||||
@@ -229,7 +229,7 @@ write_to_csv <- function(tables, path="data/csv/", create=F) {
|
|||||||
#' reading the tables from a csv is way faster than reading and repairing the data every single time
|
#' reading the tables from a csv is way faster than reading and repairing the data every single time
|
||||||
#'
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
read_from_csv <- function(path="data/csv/") {
|
read_from_csv <- function(path="inst/csv/") {
|
||||||
list(speaker = read.table(str_c(path, "speaker.csv")) %>%
|
list(speaker = read.table(str_c(path, "speaker.csv")) %>%
|
||||||
tibble() %>%
|
tibble() %>%
|
||||||
mutate(id = as.character(id)),
|
mutate(id = as.character(id)),
|
||||||
|
|||||||
+1
-1
@@ -4,7 +4,7 @@
|
|||||||
\alias{fetch_all}
|
\alias{fetch_all}
|
||||||
\title{Download available records}
|
\title{Download available records}
|
||||||
\usage{
|
\usage{
|
||||||
fetch_all(download_dir = "data/records/", create = FALSE)
|
fetch_all(download_dir = "inst/records/", create = FALSE)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{download_dir}{character}
|
\item{download_dir}{character}
|
||||||
|
|||||||
+1
-1
@@ -4,7 +4,7 @@
|
|||||||
\alias{read_all}
|
\alias{read_all}
|
||||||
\title{Parse xml records}
|
\title{Parse xml records}
|
||||||
\usage{
|
\usage{
|
||||||
read_all(path = "data/records/")
|
read_all(path = "inst/records/")
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{path}{character}
|
\item{path}{character}
|
||||||
|
|||||||
@@ -4,7 +4,7 @@
|
|||||||
\alias{read_from_csv}
|
\alias{read_from_csv}
|
||||||
\title{create a tibble from the csv file}
|
\title{create a tibble from the csv file}
|
||||||
\usage{
|
\usage{
|
||||||
read_from_csv(path = "data/csv/")
|
read_from_csv(path = "inst/csv/")
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{path}{directory to read files from
|
\item{path}{directory to read files from
|
||||||
|
|||||||
+1
-1
@@ -4,7 +4,7 @@
|
|||||||
\alias{write_to_csv}
|
\alias{write_to_csv}
|
||||||
\title{Write the parsed and repaired results into separate csv files}
|
\title{Write the parsed and repaired results into separate csv files}
|
||||||
\usage{
|
\usage{
|
||||||
write_to_csv(tables, path = "data/csv/", create = F)
|
write_to_csv(tables, path = "inst/csv/", create = F)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{tables}{list of tables to convert into a csv files.}
|
\item{tables}{list of tables to convert into a csv files.}
|
||||||
|
|||||||
@@ -26,18 +26,18 @@ library(tidyr)
|
|||||||
|
|
||||||
First, you need to download all records of the current legislative period.
|
First, you need to download all records of the current legislative period.
|
||||||
```r
|
```r
|
||||||
fetch_all("../data/records/") # path to directory where records should be stored
|
fetch_all("../inst/records/") # path to directory where records should be stored
|
||||||
```
|
```
|
||||||
Second, those `.xml` files, need to be parsed into `R` `tibbles`. This is accomplished by:
|
Second, those `.xml` files, need to be parsed into `R` `tibbles`. This is accomplished by:
|
||||||
```r
|
```r
|
||||||
read_all("../data/records/") %>% repair() -> res
|
read_all("../inst/records/") %>% repair() -> res
|
||||||
```
|
```
|
||||||
We also used `repair` to fix a bunch of formatting issues in the records and unpacked
|
We also used `repair` to fix a bunch of formatting issues in the records and unpacked
|
||||||
the result into more descriptive variables.
|
the result into more descriptive variables.
|
||||||
|
|
||||||
For development purposes, we load the tables from csv files.
|
For development purposes, we load the tables from csv files.
|
||||||
```{r}
|
```{r}
|
||||||
res <- read_from_csv('../data/csv/')
|
res <- read_from_csv('../inst/csv/')
|
||||||
```
|
```
|
||||||
and unpack our tibbles
|
and unpack our tibbles
|
||||||
```{r}
|
```{r}
|
||||||
|
|||||||
@@ -38,7 +38,7 @@ the result into more descriptive variables.
|
|||||||
|
|
||||||
For development purposes, we load the tables from csv files.
|
For development purposes, we load the tables from csv files.
|
||||||
```{r}
|
```{r}
|
||||||
res <- read_from_csv('../csv/')
|
res <- read_from_csv('../inst/csv/')
|
||||||
```
|
```
|
||||||
and unpack our tibbles
|
and unpack our tibbles
|
||||||
```{r}
|
```{r}
|
||||||
|
|||||||
@@ -25,11 +25,11 @@ library(ggplot2)
|
|||||||
|
|
||||||
First, you need to download all records of the current legislative period.
|
First, you need to download all records of the current legislative period.
|
||||||
```r
|
```r
|
||||||
fetch_all("../data/records/") # path to directory where records should be stored
|
fetch_all("../inst/records/") # path to directory where records should be stored
|
||||||
```
|
```
|
||||||
Second, those `.xml` files, need to be parsed into `R` `tibbles`. This is accomplished by:
|
Second, those `.xml` files, need to be parsed into `R` `tibbles`. This is accomplished by:
|
||||||
```r
|
```r
|
||||||
read_all("../data/records/") %>% repair() -> res
|
read_all("../inst/records/") %>% repair() -> res
|
||||||
|
|
||||||
speeches <- res$speeches
|
speeches <- res$speeches
|
||||||
speaker <- res$speaker
|
speaker <- res$speaker
|
||||||
@@ -40,7 +40,7 @@ the result into more descriptive variables.
|
|||||||
|
|
||||||
For development purposes, we load the tables from csv files.
|
For development purposes, we load the tables from csv files.
|
||||||
```{r}
|
```{r}
|
||||||
tables <- read_from_csv('../data/csv/')
|
tables <- read_from_csv('../inst/csv/')
|
||||||
|
|
||||||
comments <- tables$comments
|
comments <- tables$comments
|
||||||
speeches <- tables$speeches
|
speeches <- tables$speeches
|
||||||
@@ -50,7 +50,7 @@ talks <- tables$talks
|
|||||||
|
|
||||||
Further, we need to load a list of words that were used by Hitler but not by standard German texts.
|
Further, we need to load a list of words that were used by Hitler but not by standard German texts.
|
||||||
```{r}
|
```{r}
|
||||||
fil <- file('../data/hitler_texts/hitler_words')
|
fil <- file('../inst/hitler_texts/hitler_words')
|
||||||
Worte <- readLines(fil)
|
Worte <- readLines(fil)
|
||||||
hitlerwords <- tibble(Worte)
|
hitlerwords <- tibble(Worte)
|
||||||
```
|
```
|
||||||
|
|||||||
Reference in New Issue
Block a user