Compare commits
74
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9d456bfa60 | ||
|
|
62fe9d497d | ||
|
|
29a7974941 | ||
|
|
e31ccabf18 | ||
|
|
622fd4db07 | ||
|
|
6aa80534f8 | ||
|
|
605e5e976a | ||
|
|
8e691e5d11 | ||
|
|
6b1f8a64b2 | ||
|
|
7e304d12bb | ||
|
|
864c0016cc | ||
|
|
489254dadf | ||
|
|
05755f9737 | ||
|
|
01ec0de76f | ||
|
|
fbfbed3900 | ||
|
|
b09742815a | ||
|
|
b11289e093 | ||
|
|
7daf9e553c | ||
|
|
d91403f0c8 | ||
|
|
d657ca3fbe | ||
|
|
c53d842a1e | ||
|
|
12452df224 | ||
|
|
f753920d34 | ||
|
|
77f79f0d86 | ||
|
|
c80ec6259e | ||
|
|
8e2a9bd399 | ||
|
|
e038d86339 | ||
|
|
4093cb603c | ||
|
|
1152115a0f | ||
|
|
4dbf90127a | ||
|
|
2a9035e630 | ||
|
|
5490f9fed6 | ||
|
|
4649658fa7 | ||
|
|
b02ab91c31 | ||
|
|
03f8ca0813 | ||
|
|
4d00f4d1b0 | ||
|
|
545e5c9bab | ||
|
|
1db9e3f59a | ||
|
|
1546f84d80 | ||
|
|
0597f68694 | ||
|
|
d5f74c5bf3 | ||
|
|
0f07b002e6 | ||
|
|
789bf57756 | ||
|
|
502dc45781 | ||
|
|
4b3e85be62 | ||
|
|
feed583fa9 | ||
|
|
53fdb7530b | ||
|
|
bf30511678 | ||
|
|
5e9f70627d | ||
|
|
d5da636020 | ||
|
|
1433159e09 | ||
|
|
c22bc0b91b | ||
|
|
5f9343bf7f | ||
|
|
a0df02dbed | ||
|
|
7315dd8793 | ||
|
|
5dc308c16e | ||
|
|
7771b2ebd8 | ||
|
|
f91456116f | ||
|
|
64ede5c757 | ||
|
|
09f5e5da0d | ||
|
|
610982427a | ||
|
|
e2caf1ff4e | ||
|
|
354d18050b | ||
|
|
7dd136b869 | ||
|
|
834a840734 | ||
|
|
0a97674b51 | ||
|
|
ef29269d45 | ||
|
|
6e8e942030 | ||
|
|
ccb0649cb9 | ||
|
|
7067877584 | ||
|
|
002ee56853 | ||
|
|
4fe45feec9 | ||
|
|
75042d7128 | ||
|
|
88a62a22d7 |
@@ -2,3 +2,4 @@
|
|||||||
^Meta$
|
^Meta$
|
||||||
^.*\.Rproj$
|
^.*\.Rproj$
|
||||||
^\.Rproj\.user$
|
^\.Rproj\.user$
|
||||||
|
^LICENSE\.md$
|
||||||
|
|||||||
+6
-4
@@ -1,10 +1,12 @@
|
|||||||
*.xml
|
*.xml
|
||||||
/doc/
|
/doc/
|
||||||
/Meta/
|
/Meta/
|
||||||
/reports/
|
/inst/reports/
|
||||||
!/reports/*.pdf
|
!/inst/reports/*.pdf
|
||||||
!/reports/*.tex
|
!/inst/reports/*.tex
|
||||||
/csv/*
|
/inst/csv/*
|
||||||
/parlament_49_53_texts/
|
/parlament_49_53_texts/
|
||||||
.Rproj.user
|
.Rproj.user
|
||||||
*.Rproj
|
*.Rproj
|
||||||
|
*.RData
|
||||||
|
*.Rhistory
|
||||||
|
|||||||
+19
-9
@@ -1,26 +1,36 @@
|
|||||||
Package: hateimparlament
|
Package: hateimparlament
|
||||||
Title: Protocolanalysis of German Bundestag
|
Title: Recordanalysis Of Bundestag
|
||||||
Version: 0.0.0.9000
|
Version: 0.0.0.9000
|
||||||
Authors@R:
|
Authors@R: c(
|
||||||
person(given = "First",
|
person(given = "Leon",
|
||||||
family = "Last",
|
family = "Burgard",
|
||||||
|
role = c("aut")),
|
||||||
|
person(given = "Josua",
|
||||||
|
family = "Kugler",
|
||||||
|
role = c("aut")),
|
||||||
|
person(given = "Christian",
|
||||||
|
family = "Merten",
|
||||||
role = c("aut", "cre"),
|
role = c("aut", "cre"),
|
||||||
email = "first.last@example.com",
|
email = "christian@merten.dev"))
|
||||||
comment = c(ORCID = "YOUR-ORCID-ID"))
|
Description: Downloads, parses and analyses parliamentary records of the 19th legislative
|
||||||
Description: Downloads, parses and analyses protocols of the current German parliament (Bundestag).
|
period of the German parliament (Bundestag).
|
||||||
License: `use_mit_license()`, `use_gpl3_license()` or friends to pick a
|
URL: https://git.flavigny.de/christian/hateimparlament
|
||||||
license
|
BugReports: https://git.flavigny.de/christian/hateimparlament/issues
|
||||||
|
License: GPL (>= 3)
|
||||||
Encoding: UTF-8
|
Encoding: UTF-8
|
||||||
LazyData: true
|
LazyData: true
|
||||||
Roxygen: list(markdown = TRUE)
|
Roxygen: list(markdown = TRUE)
|
||||||
RoxygenNote: 7.1.1
|
RoxygenNote: 7.1.1
|
||||||
Imports:
|
Imports:
|
||||||
dplyr,
|
dplyr,
|
||||||
|
lubridate,
|
||||||
pbapply,
|
pbapply,
|
||||||
purrr,
|
purrr,
|
||||||
|
rlang,
|
||||||
rvest,
|
rvest,
|
||||||
stringr,
|
stringr,
|
||||||
tibble,
|
tibble,
|
||||||
|
tidyr,
|
||||||
xml2
|
xml2
|
||||||
Suggests:
|
Suggests:
|
||||||
rmarkdown,
|
rmarkdown,
|
||||||
|
|||||||
+595
@@ -0,0 +1,595 @@
|
|||||||
|
GNU General Public License
|
||||||
|
==========================
|
||||||
|
|
||||||
|
_Version 3, 29 June 2007_
|
||||||
|
_Copyright © 2007 Free Software Foundation, Inc. <<http://fsf.org/>>_
|
||||||
|
|
||||||
|
Everyone is permitted to copy and distribute verbatim copies of this license
|
||||||
|
document, but changing it is not allowed.
|
||||||
|
|
||||||
|
## Preamble
|
||||||
|
|
||||||
|
The GNU General Public License is a free, copyleft license for software and other
|
||||||
|
kinds of works.
|
||||||
|
|
||||||
|
The licenses for most software and other practical works are designed to take away
|
||||||
|
your freedom to share and change the works. By contrast, the GNU General Public
|
||||||
|
License is intended to guarantee your freedom to share and change all versions of a
|
||||||
|
program--to make sure it remains free software for all its users. We, the Free
|
||||||
|
Software Foundation, use the GNU General Public License for most of our software; it
|
||||||
|
applies also to any other work released this way by its authors. You can apply it to
|
||||||
|
your programs, too.
|
||||||
|
|
||||||
|
When we speak of free software, we are referring to freedom, not price. Our General
|
||||||
|
Public Licenses are designed to make sure that you have the freedom to distribute
|
||||||
|
copies of free software (and charge for them if you wish), that you receive source
|
||||||
|
code or can get it if you want it, that you can change the software or use pieces of
|
||||||
|
it in new free programs, and that you know you can do these things.
|
||||||
|
|
||||||
|
To protect your rights, we need to prevent others from denying you these rights or
|
||||||
|
asking you to surrender the rights. Therefore, you have certain responsibilities if
|
||||||
|
you distribute copies of the software, or if you modify it: responsibilities to
|
||||||
|
respect the freedom of others.
|
||||||
|
|
||||||
|
For example, if you distribute copies of such a program, whether gratis or for a fee,
|
||||||
|
you must pass on to the recipients the same freedoms that you received. You must make
|
||||||
|
sure that they, too, receive or can get the source code. And you must show them these
|
||||||
|
terms so they know their rights.
|
||||||
|
|
||||||
|
Developers that use the GNU GPL protect your rights with two steps: **(1)** assert
|
||||||
|
copyright on the software, and **(2)** offer you this License giving you legal permission
|
||||||
|
to copy, distribute and/or modify it.
|
||||||
|
|
||||||
|
For the developers' and authors' protection, the GPL clearly explains that there is
|
||||||
|
no warranty for this free software. For both users' and authors' sake, the GPL
|
||||||
|
requires that modified versions be marked as changed, so that their problems will not
|
||||||
|
be attributed erroneously to authors of previous versions.
|
||||||
|
|
||||||
|
Some devices are designed to deny users access to install or run modified versions of
|
||||||
|
the software inside them, although the manufacturer can do so. This is fundamentally
|
||||||
|
incompatible with the aim of protecting users' freedom to change the software. The
|
||||||
|
systematic pattern of such abuse occurs in the area of products for individuals to
|
||||||
|
use, which is precisely where it is most unacceptable. Therefore, we have designed
|
||||||
|
this version of the GPL to prohibit the practice for those products. If such problems
|
||||||
|
arise substantially in other domains, we stand ready to extend this provision to
|
||||||
|
those domains in future versions of the GPL, as needed to protect the freedom of
|
||||||
|
users.
|
||||||
|
|
||||||
|
Finally, every program is threatened constantly by software patents. States should
|
||||||
|
not allow patents to restrict development and use of software on general-purpose
|
||||||
|
computers, but in those that do, we wish to avoid the special danger that patents
|
||||||
|
applied to a free program could make it effectively proprietary. To prevent this, the
|
||||||
|
GPL assures that patents cannot be used to render the program non-free.
|
||||||
|
|
||||||
|
The precise terms and conditions for copying, distribution and modification follow.
|
||||||
|
|
||||||
|
## TERMS AND CONDITIONS
|
||||||
|
|
||||||
|
### 0. Definitions
|
||||||
|
|
||||||
|
“This License” refers to version 3 of the GNU General Public License.
|
||||||
|
|
||||||
|
“Copyright” also means copyright-like laws that apply to other kinds of
|
||||||
|
works, such as semiconductor masks.
|
||||||
|
|
||||||
|
“The Program” refers to any copyrightable work licensed under this
|
||||||
|
License. Each licensee is addressed as “you”. “Licensees” and
|
||||||
|
“recipients” may be individuals or organizations.
|
||||||
|
|
||||||
|
To “modify” a work means to copy from or adapt all or part of the work in
|
||||||
|
a fashion requiring copyright permission, other than the making of an exact copy. The
|
||||||
|
resulting work is called a “modified version” of the earlier work or a
|
||||||
|
work “based on” the earlier work.
|
||||||
|
|
||||||
|
A “covered work” means either the unmodified Program or a work based on
|
||||||
|
the Program.
|
||||||
|
|
||||||
|
To “propagate” a work means to do anything with it that, without
|
||||||
|
permission, would make you directly or secondarily liable for infringement under
|
||||||
|
applicable copyright law, except executing it on a computer or modifying a private
|
||||||
|
copy. Propagation includes copying, distribution (with or without modification),
|
||||||
|
making available to the public, and in some countries other activities as well.
|
||||||
|
|
||||||
|
To “convey” a work means any kind of propagation that enables other
|
||||||
|
parties to make or receive copies. Mere interaction with a user through a computer
|
||||||
|
network, with no transfer of a copy, is not conveying.
|
||||||
|
|
||||||
|
An interactive user interface displays “Appropriate Legal Notices” to the
|
||||||
|
extent that it includes a convenient and prominently visible feature that **(1)**
|
||||||
|
displays an appropriate copyright notice, and **(2)** tells the user that there is no
|
||||||
|
warranty for the work (except to the extent that warranties are provided), that
|
||||||
|
licensees may convey the work under this License, and how to view a copy of this
|
||||||
|
License. If the interface presents a list of user commands or options, such as a
|
||||||
|
menu, a prominent item in the list meets this criterion.
|
||||||
|
|
||||||
|
### 1. Source Code
|
||||||
|
|
||||||
|
The “source code” for a work means the preferred form of the work for
|
||||||
|
making modifications to it. “Object code” means any non-source form of a
|
||||||
|
work.
|
||||||
|
|
||||||
|
A “Standard Interface” means an interface that either is an official
|
||||||
|
standard defined by a recognized standards body, or, in the case of interfaces
|
||||||
|
specified for a particular programming language, one that is widely used among
|
||||||
|
developers working in that language.
|
||||||
|
|
||||||
|
The “System Libraries” of an executable work include anything, other than
|
||||||
|
the work as a whole, that **(a)** is included in the normal form of packaging a Major
|
||||||
|
Component, but which is not part of that Major Component, and **(b)** serves only to
|
||||||
|
enable use of the work with that Major Component, or to implement a Standard
|
||||||
|
Interface for which an implementation is available to the public in source code form.
|
||||||
|
A “Major Component”, in this context, means a major essential component
|
||||||
|
(kernel, window system, and so on) of the specific operating system (if any) on which
|
||||||
|
the executable work runs, or a compiler used to produce the work, or an object code
|
||||||
|
interpreter used to run it.
|
||||||
|
|
||||||
|
The “Corresponding Source” for a work in object code form means all the
|
||||||
|
source code needed to generate, install, and (for an executable work) run the object
|
||||||
|
code and to modify the work, including scripts to control those activities. However,
|
||||||
|
it does not include the work's System Libraries, or general-purpose tools or
|
||||||
|
generally available free programs which are used unmodified in performing those
|
||||||
|
activities but which are not part of the work. For example, Corresponding Source
|
||||||
|
includes interface definition files associated with source files for the work, and
|
||||||
|
the source code for shared libraries and dynamically linked subprograms that the work
|
||||||
|
is specifically designed to require, such as by intimate data communication or
|
||||||
|
control flow between those subprograms and other parts of the work.
|
||||||
|
|
||||||
|
The Corresponding Source need not include anything that users can regenerate
|
||||||
|
automatically from other parts of the Corresponding Source.
|
||||||
|
|
||||||
|
The Corresponding Source for a work in source code form is that same work.
|
||||||
|
|
||||||
|
### 2. Basic Permissions
|
||||||
|
|
||||||
|
All rights granted under this License are granted for the term of copyright on the
|
||||||
|
Program, and are irrevocable provided the stated conditions are met. This License
|
||||||
|
explicitly affirms your unlimited permission to run the unmodified Program. The
|
||||||
|
output from running a covered work is covered by this License only if the output,
|
||||||
|
given its content, constitutes a covered work. This License acknowledges your rights
|
||||||
|
of fair use or other equivalent, as provided by copyright law.
|
||||||
|
|
||||||
|
You may make, run and propagate covered works that you do not convey, without
|
||||||
|
conditions so long as your license otherwise remains in force. You may convey covered
|
||||||
|
works to others for the sole purpose of having them make modifications exclusively
|
||||||
|
for you, or provide you with facilities for running those works, provided that you
|
||||||
|
comply with the terms of this License in conveying all material for which you do not
|
||||||
|
control copyright. Those thus making or running the covered works for you must do so
|
||||||
|
exclusively on your behalf, under your direction and control, on terms that prohibit
|
||||||
|
them from making any copies of your copyrighted material outside their relationship
|
||||||
|
with you.
|
||||||
|
|
||||||
|
Conveying under any other circumstances is permitted solely under the conditions
|
||||||
|
stated below. Sublicensing is not allowed; section 10 makes it unnecessary.
|
||||||
|
|
||||||
|
### 3. Protecting Users' Legal Rights From Anti-Circumvention Law
|
||||||
|
|
||||||
|
No covered work shall be deemed part of an effective technological measure under any
|
||||||
|
applicable law fulfilling obligations under article 11 of the WIPO copyright treaty
|
||||||
|
adopted on 20 December 1996, or similar laws prohibiting or restricting circumvention
|
||||||
|
of such measures.
|
||||||
|
|
||||||
|
When you convey a covered work, you waive any legal power to forbid circumvention of
|
||||||
|
technological measures to the extent such circumvention is effected by exercising
|
||||||
|
rights under this License with respect to the covered work, and you disclaim any
|
||||||
|
intention to limit operation or modification of the work as a means of enforcing,
|
||||||
|
against the work's users, your or third parties' legal rights to forbid circumvention
|
||||||
|
of technological measures.
|
||||||
|
|
||||||
|
### 4. Conveying Verbatim Copies
|
||||||
|
|
||||||
|
You may convey verbatim copies of the Program's source code as you receive it, in any
|
||||||
|
medium, provided that you conspicuously and appropriately publish on each copy an
|
||||||
|
appropriate copyright notice; keep intact all notices stating that this License and
|
||||||
|
any non-permissive terms added in accord with section 7 apply to the code; keep
|
||||||
|
intact all notices of the absence of any warranty; and give all recipients a copy of
|
||||||
|
this License along with the Program.
|
||||||
|
|
||||||
|
You may charge any price or no price for each copy that you convey, and you may offer
|
||||||
|
support or warranty protection for a fee.
|
||||||
|
|
||||||
|
### 5. Conveying Modified Source Versions
|
||||||
|
|
||||||
|
You may convey a work based on the Program, or the modifications to produce it from
|
||||||
|
the Program, in the form of source code under the terms of section 4, provided that
|
||||||
|
you also meet all of these conditions:
|
||||||
|
|
||||||
|
* **a)** The work must carry prominent notices stating that you modified it, and giving a
|
||||||
|
relevant date.
|
||||||
|
* **b)** The work must carry prominent notices stating that it is released under this
|
||||||
|
License and any conditions added under section 7. This requirement modifies the
|
||||||
|
requirement in section 4 to “keep intact all notices”.
|
||||||
|
* **c)** You must license the entire work, as a whole, under this License to anyone who
|
||||||
|
comes into possession of a copy. This License will therefore apply, along with any
|
||||||
|
applicable section 7 additional terms, to the whole of the work, and all its parts,
|
||||||
|
regardless of how they are packaged. This License gives no permission to license the
|
||||||
|
work in any other way, but it does not invalidate such permission if you have
|
||||||
|
separately received it.
|
||||||
|
* **d)** If the work has interactive user interfaces, each must display Appropriate Legal
|
||||||
|
Notices; however, if the Program has interactive interfaces that do not display
|
||||||
|
Appropriate Legal Notices, your work need not make them do so.
|
||||||
|
|
||||||
|
A compilation of a covered work with other separate and independent works, which are
|
||||||
|
not by their nature extensions of the covered work, and which are not combined with
|
||||||
|
it such as to form a larger program, in or on a volume of a storage or distribution
|
||||||
|
medium, is called an “aggregate” if the compilation and its resulting
|
||||||
|
copyright are not used to limit the access or legal rights of the compilation's users
|
||||||
|
beyond what the individual works permit. Inclusion of a covered work in an aggregate
|
||||||
|
does not cause this License to apply to the other parts of the aggregate.
|
||||||
|
|
||||||
|
### 6. Conveying Non-Source Forms
|
||||||
|
|
||||||
|
You may convey a covered work in object code form under the terms of sections 4 and
|
||||||
|
5, provided that you also convey the machine-readable Corresponding Source under the
|
||||||
|
terms of this License, in one of these ways:
|
||||||
|
|
||||||
|
* **a)** Convey the object code in, or embodied in, a physical product (including a
|
||||||
|
physical distribution medium), accompanied by the Corresponding Source fixed on a
|
||||||
|
durable physical medium customarily used for software interchange.
|
||||||
|
* **b)** Convey the object code in, or embodied in, a physical product (including a
|
||||||
|
physical distribution medium), accompanied by a written offer, valid for at least
|
||||||
|
three years and valid for as long as you offer spare parts or customer support for
|
||||||
|
that product model, to give anyone who possesses the object code either **(1)** a copy of
|
||||||
|
the Corresponding Source for all the software in the product that is covered by this
|
||||||
|
License, on a durable physical medium customarily used for software interchange, for
|
||||||
|
a price no more than your reasonable cost of physically performing this conveying of
|
||||||
|
source, or **(2)** access to copy the Corresponding Source from a network server at no
|
||||||
|
charge.
|
||||||
|
* **c)** Convey individual copies of the object code with a copy of the written offer to
|
||||||
|
provide the Corresponding Source. This alternative is allowed only occasionally and
|
||||||
|
noncommercially, and only if you received the object code with such an offer, in
|
||||||
|
accord with subsection 6b.
|
||||||
|
* **d)** Convey the object code by offering access from a designated place (gratis or for
|
||||||
|
a charge), and offer equivalent access to the Corresponding Source in the same way
|
||||||
|
through the same place at no further charge. You need not require recipients to copy
|
||||||
|
the Corresponding Source along with the object code. If the place to copy the object
|
||||||
|
code is a network server, the Corresponding Source may be on a different server
|
||||||
|
(operated by you or a third party) that supports equivalent copying facilities,
|
||||||
|
provided you maintain clear directions next to the object code saying where to find
|
||||||
|
the Corresponding Source. Regardless of what server hosts the Corresponding Source,
|
||||||
|
you remain obligated to ensure that it is available for as long as needed to satisfy
|
||||||
|
these requirements.
|
||||||
|
* **e)** Convey the object code using peer-to-peer transmission, provided you inform
|
||||||
|
other peers where the object code and Corresponding Source of the work are being
|
||||||
|
offered to the general public at no charge under subsection 6d.
|
||||||
|
|
||||||
|
A separable portion of the object code, whose source code is excluded from the
|
||||||
|
Corresponding Source as a System Library, need not be included in conveying the
|
||||||
|
object code work.
|
||||||
|
|
||||||
|
A “User Product” is either **(1)** a “consumer product”, which
|
||||||
|
means any tangible personal property which is normally used for personal, family, or
|
||||||
|
household purposes, or **(2)** anything designed or sold for incorporation into a
|
||||||
|
dwelling. In determining whether a product is a consumer product, doubtful cases
|
||||||
|
shall be resolved in favor of coverage. For a particular product received by a
|
||||||
|
particular user, “normally used” refers to a typical or common use of
|
||||||
|
that class of product, regardless of the status of the particular user or of the way
|
||||||
|
in which the particular user actually uses, or expects or is expected to use, the
|
||||||
|
product. A product is a consumer product regardless of whether the product has
|
||||||
|
substantial commercial, industrial or non-consumer uses, unless such uses represent
|
||||||
|
the only significant mode of use of the product.
|
||||||
|
|
||||||
|
“Installation Information” for a User Product means any methods,
|
||||||
|
procedures, authorization keys, or other information required to install and execute
|
||||||
|
modified versions of a covered work in that User Product from a modified version of
|
||||||
|
its Corresponding Source. The information must suffice to ensure that the continued
|
||||||
|
functioning of the modified object code is in no case prevented or interfered with
|
||||||
|
solely because modification has been made.
|
||||||
|
|
||||||
|
If you convey an object code work under this section in, or with, or specifically for
|
||||||
|
use in, a User Product, and the conveying occurs as part of a transaction in which
|
||||||
|
the right of possession and use of the User Product is transferred to the recipient
|
||||||
|
in perpetuity or for a fixed term (regardless of how the transaction is
|
||||||
|
characterized), the Corresponding Source conveyed under this section must be
|
||||||
|
accompanied by the Installation Information. But this requirement does not apply if
|
||||||
|
neither you nor any third party retains the ability to install modified object code
|
||||||
|
on the User Product (for example, the work has been installed in ROM).
|
||||||
|
|
||||||
|
The requirement to provide Installation Information does not include a requirement to
|
||||||
|
continue to provide support service, warranty, or updates for a work that has been
|
||||||
|
modified or installed by the recipient, or for the User Product in which it has been
|
||||||
|
modified or installed. Access to a network may be denied when the modification itself
|
||||||
|
materially and adversely affects the operation of the network or violates the rules
|
||||||
|
and protocols for communication across the network.
|
||||||
|
|
||||||
|
Corresponding Source conveyed, and Installation Information provided, in accord with
|
||||||
|
this section must be in a format that is publicly documented (and with an
|
||||||
|
implementation available to the public in source code form), and must require no
|
||||||
|
special password or key for unpacking, reading or copying.
|
||||||
|
|
||||||
|
### 7. Additional Terms
|
||||||
|
|
||||||
|
“Additional permissions” are terms that supplement the terms of this
|
||||||
|
License by making exceptions from one or more of its conditions. Additional
|
||||||
|
permissions that are applicable to the entire Program shall be treated as though they
|
||||||
|
were included in this License, to the extent that they are valid under applicable
|
||||||
|
law. If additional permissions apply only to part of the Program, that part may be
|
||||||
|
used separately under those permissions, but the entire Program remains governed by
|
||||||
|
this License without regard to the additional permissions.
|
||||||
|
|
||||||
|
When you convey a copy of a covered work, you may at your option remove any
|
||||||
|
additional permissions from that copy, or from any part of it. (Additional
|
||||||
|
permissions may be written to require their own removal in certain cases when you
|
||||||
|
modify the work.) You may place additional permissions on material, added by you to a
|
||||||
|
covered work, for which you have or can give appropriate copyright permission.
|
||||||
|
|
||||||
|
Notwithstanding any other provision of this License, for material you add to a
|
||||||
|
covered work, you may (if authorized by the copyright holders of that material)
|
||||||
|
supplement the terms of this License with terms:
|
||||||
|
|
||||||
|
* **a)** Disclaiming warranty or limiting liability differently from the terms of
|
||||||
|
sections 15 and 16 of this License; or
|
||||||
|
* **b)** Requiring preservation of specified reasonable legal notices or author
|
||||||
|
attributions in that material or in the Appropriate Legal Notices displayed by works
|
||||||
|
containing it; or
|
||||||
|
* **c)** Prohibiting misrepresentation of the origin of that material, or requiring that
|
||||||
|
modified versions of such material be marked in reasonable ways as different from the
|
||||||
|
original version; or
|
||||||
|
* **d)** Limiting the use for publicity purposes of names of licensors or authors of the
|
||||||
|
material; or
|
||||||
|
* **e)** Declining to grant rights under trademark law for use of some trade names,
|
||||||
|
trademarks, or service marks; or
|
||||||
|
* **f)** Requiring indemnification of licensors and authors of that material by anyone
|
||||||
|
who conveys the material (or modified versions of it) with contractual assumptions of
|
||||||
|
liability to the recipient, for any liability that these contractual assumptions
|
||||||
|
directly impose on those licensors and authors.
|
||||||
|
|
||||||
|
All other non-permissive additional terms are considered “further
|
||||||
|
restrictions” within the meaning of section 10. If the Program as you received
|
||||||
|
it, or any part of it, contains a notice stating that it is governed by this License
|
||||||
|
along with a term that is a further restriction, you may remove that term. If a
|
||||||
|
license document contains a further restriction but permits relicensing or conveying
|
||||||
|
under this License, you may add to a covered work material governed by the terms of
|
||||||
|
that license document, provided that the further restriction does not survive such
|
||||||
|
relicensing or conveying.
|
||||||
|
|
||||||
|
If you add terms to a covered work in accord with this section, you must place, in
|
||||||
|
the relevant source files, a statement of the additional terms that apply to those
|
||||||
|
files, or a notice indicating where to find the applicable terms.
|
||||||
|
|
||||||
|
Additional terms, permissive or non-permissive, may be stated in the form of a
|
||||||
|
separately written license, or stated as exceptions; the above requirements apply
|
||||||
|
either way.
|
||||||
|
|
||||||
|
### 8. Termination
|
||||||
|
|
||||||
|
You may not propagate or modify a covered work except as expressly provided under
|
||||||
|
this License. Any attempt otherwise to propagate or modify it is void, and will
|
||||||
|
automatically terminate your rights under this License (including any patent licenses
|
||||||
|
granted under the third paragraph of section 11).
|
||||||
|
|
||||||
|
However, if you cease all violation of this License, then your license from a
|
||||||
|
particular copyright holder is reinstated **(a)** provisionally, unless and until the
|
||||||
|
copyright holder explicitly and finally terminates your license, and **(b)** permanently,
|
||||||
|
if the copyright holder fails to notify you of the violation by some reasonable means
|
||||||
|
prior to 60 days after the cessation.
|
||||||
|
|
||||||
|
Moreover, your license from a particular copyright holder is reinstated permanently
|
||||||
|
if the copyright holder notifies you of the violation by some reasonable means, this
|
||||||
|
is the first time you have received notice of violation of this License (for any
|
||||||
|
work) from that copyright holder, and you cure the violation prior to 30 days after
|
||||||
|
your receipt of the notice.
|
||||||
|
|
||||||
|
Termination of your rights under this section does not terminate the licenses of
|
||||||
|
parties who have received copies or rights from you under this License. If your
|
||||||
|
rights have been terminated and not permanently reinstated, you do not qualify to
|
||||||
|
receive new licenses for the same material under section 10.
|
||||||
|
|
||||||
|
### 9. Acceptance Not Required for Having Copies
|
||||||
|
|
||||||
|
You are not required to accept this License in order to receive or run a copy of the
|
||||||
|
Program. Ancillary propagation of a covered work occurring solely as a consequence of
|
||||||
|
using peer-to-peer transmission to receive a copy likewise does not require
|
||||||
|
acceptance. However, nothing other than this License grants you permission to
|
||||||
|
propagate or modify any covered work. These actions infringe copyright if you do not
|
||||||
|
accept this License. Therefore, by modifying or propagating a covered work, you
|
||||||
|
indicate your acceptance of this License to do so.
|
||||||
|
|
||||||
|
### 10. Automatic Licensing of Downstream Recipients
|
||||||
|
|
||||||
|
Each time you convey a covered work, the recipient automatically receives a license
|
||||||
|
from the original licensors, to run, modify and propagate that work, subject to this
|
||||||
|
License. You are not responsible for enforcing compliance by third parties with this
|
||||||
|
License.
|
||||||
|
|
||||||
|
An “entity transaction” is a transaction transferring control of an
|
||||||
|
organization, or substantially all assets of one, or subdividing an organization, or
|
||||||
|
merging organizations. If propagation of a covered work results from an entity
|
||||||
|
transaction, each party to that transaction who receives a copy of the work also
|
||||||
|
receives whatever licenses to the work the party's predecessor in interest had or
|
||||||
|
could give under the previous paragraph, plus a right to possession of the
|
||||||
|
Corresponding Source of the work from the predecessor in interest, if the predecessor
|
||||||
|
has it or can get it with reasonable efforts.
|
||||||
|
|
||||||
|
You may not impose any further restrictions on the exercise of the rights granted or
|
||||||
|
affirmed under this License. For example, you may not impose a license fee, royalty,
|
||||||
|
or other charge for exercise of rights granted under this License, and you may not
|
||||||
|
initiate litigation (including a cross-claim or counterclaim in a lawsuit) alleging
|
||||||
|
that any patent claim is infringed by making, using, selling, offering for sale, or
|
||||||
|
importing the Program or any portion of it.
|
||||||
|
|
||||||
|
### 11. Patents
|
||||||
|
|
||||||
|
A “contributor” is a copyright holder who authorizes use under this
|
||||||
|
License of the Program or a work on which the Program is based. The work thus
|
||||||
|
licensed is called the contributor's “contributor version”.
|
||||||
|
|
||||||
|
A contributor's “essential patent claims” are all patent claims owned or
|
||||||
|
controlled by the contributor, whether already acquired or hereafter acquired, that
|
||||||
|
would be infringed by some manner, permitted by this License, of making, using, or
|
||||||
|
selling its contributor version, but do not include claims that would be infringed
|
||||||
|
only as a consequence of further modification of the contributor version. For
|
||||||
|
purposes of this definition, “control” includes the right to grant patent
|
||||||
|
sublicenses in a manner consistent with the requirements of this License.
|
||||||
|
|
||||||
|
Each contributor grants you a non-exclusive, worldwide, royalty-free patent license
|
||||||
|
under the contributor's essential patent claims, to make, use, sell, offer for sale,
|
||||||
|
import and otherwise run, modify and propagate the contents of its contributor
|
||||||
|
version.
|
||||||
|
|
||||||
|
In the following three paragraphs, a “patent license” is any express
|
||||||
|
agreement or commitment, however denominated, not to enforce a patent (such as an
|
||||||
|
express permission to practice a patent or covenant not to sue for patent
|
||||||
|
infringement). To “grant” such a patent license to a party means to make
|
||||||
|
such an agreement or commitment not to enforce a patent against the party.
|
||||||
|
|
||||||
|
If you convey a covered work, knowingly relying on a patent license, and the
|
||||||
|
Corresponding Source of the work is not available for anyone to copy, free of charge
|
||||||
|
and under the terms of this License, through a publicly available network server or
|
||||||
|
other readily accessible means, then you must either **(1)** cause the Corresponding
|
||||||
|
Source to be so available, or **(2)** arrange to deprive yourself of the benefit of the
|
||||||
|
patent license for this particular work, or **(3)** arrange, in a manner consistent with
|
||||||
|
the requirements of this License, to extend the patent license to downstream
|
||||||
|
recipients. “Knowingly relying” means you have actual knowledge that, but
|
||||||
|
for the patent license, your conveying the covered work in a country, or your
|
||||||
|
recipient's use of the covered work in a country, would infringe one or more
|
||||||
|
identifiable patents in that country that you have reason to believe are valid.
|
||||||
|
|
||||||
|
If, pursuant to or in connection with a single transaction or arrangement, you
|
||||||
|
convey, or propagate by procuring conveyance of, a covered work, and grant a patent
|
||||||
|
license to some of the parties receiving the covered work authorizing them to use,
|
||||||
|
propagate, modify or convey a specific copy of the covered work, then the patent
|
||||||
|
license you grant is automatically extended to all recipients of the covered work and
|
||||||
|
works based on it.
|
||||||
|
|
||||||
|
A patent license is “discriminatory” if it does not include within the
|
||||||
|
scope of its coverage, prohibits the exercise of, or is conditioned on the
|
||||||
|
non-exercise of one or more of the rights that are specifically granted under this
|
||||||
|
License. You may not convey a covered work if you are a party to an arrangement with
|
||||||
|
a third party that is in the business of distributing software, under which you make
|
||||||
|
payment to the third party based on the extent of your activity of conveying the
|
||||||
|
work, and under which the third party grants, to any of the parties who would receive
|
||||||
|
the covered work from you, a discriminatory patent license **(a)** in connection with
|
||||||
|
copies of the covered work conveyed by you (or copies made from those copies), or **(b)**
|
||||||
|
primarily for and in connection with specific products or compilations that contain
|
||||||
|
the covered work, unless you entered into that arrangement, or that patent license
|
||||||
|
was granted, prior to 28 March 2007.
|
||||||
|
|
||||||
|
Nothing in this License shall be construed as excluding or limiting any implied
|
||||||
|
license or other defenses to infringement that may otherwise be available to you
|
||||||
|
under applicable patent law.
|
||||||
|
|
||||||
|
### 12. No Surrender of Others' Freedom
|
||||||
|
|
||||||
|
If conditions are imposed on you (whether by court order, agreement or otherwise)
|
||||||
|
that contradict the conditions of this License, they do not excuse you from the
|
||||||
|
conditions of this License. If you cannot convey a covered work so as to satisfy
|
||||||
|
simultaneously your obligations under this License and any other pertinent
|
||||||
|
obligations, then as a consequence you may not convey it at all. For example, if you
|
||||||
|
agree to terms that obligate you to collect a royalty for further conveying from
|
||||||
|
those to whom you convey the Program, the only way you could satisfy both those terms
|
||||||
|
and this License would be to refrain entirely from conveying the Program.
|
||||||
|
|
||||||
|
### 13. Use with the GNU Affero General Public License
|
||||||
|
|
||||||
|
Notwithstanding any other provision of this License, you have permission to link or
|
||||||
|
combine any covered work with a work licensed under version 3 of the GNU Affero
|
||||||
|
General Public License into a single combined work, and to convey the resulting work.
|
||||||
|
The terms of this License will continue to apply to the part which is the covered
|
||||||
|
work, but the special requirements of the GNU Affero General Public License, section
|
||||||
|
13, concerning interaction through a network will apply to the combination as such.
|
||||||
|
|
||||||
|
### 14. Revised Versions of this License
|
||||||
|
|
||||||
|
The Free Software Foundation may publish revised and/or new versions of the GNU
|
||||||
|
General Public License from time to time. Such new versions will be similar in spirit
|
||||||
|
to the present version, but may differ in detail to address new problems or concerns.
|
||||||
|
|
||||||
|
Each version is given a distinguishing version number. If the Program specifies that
|
||||||
|
a certain numbered version of the GNU General Public License “or any later
|
||||||
|
version” applies to it, you have the option of following the terms and
|
||||||
|
conditions either of that numbered version or of any later version published by the
|
||||||
|
Free Software Foundation. If the Program does not specify a version number of the GNU
|
||||||
|
General Public License, you may choose any version ever published by the Free
|
||||||
|
Software Foundation.
|
||||||
|
|
||||||
|
If the Program specifies that a proxy can decide which future versions of the GNU
|
||||||
|
General Public License can be used, that proxy's public statement of acceptance of a
|
||||||
|
version permanently authorizes you to choose that version for the Program.
|
||||||
|
|
||||||
|
Later license versions may give you additional or different permissions. However, no
|
||||||
|
additional obligations are imposed on any author or copyright holder as a result of
|
||||||
|
your choosing to follow a later version.
|
||||||
|
|
||||||
|
### 15. Disclaimer of Warranty
|
||||||
|
|
||||||
|
THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY APPLICABLE LAW.
|
||||||
|
EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT HOLDERS AND/OR OTHER PARTIES
|
||||||
|
PROVIDE THE PROGRAM “AS IS” WITHOUT WARRANTY OF ANY KIND, EITHER
|
||||||
|
EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED WARRANTIES OF
|
||||||
|
MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE. THE ENTIRE RISK AS TO THE
|
||||||
|
QUALITY AND PERFORMANCE OF THE PROGRAM IS WITH YOU. SHOULD THE PROGRAM PROVE
|
||||||
|
DEFECTIVE, YOU ASSUME THE COST OF ALL NECESSARY SERVICING, REPAIR OR CORRECTION.
|
||||||
|
|
||||||
|
### 16. Limitation of Liability
|
||||||
|
|
||||||
|
IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING WILL ANY
|
||||||
|
COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS THE PROGRAM AS
|
||||||
|
PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY GENERAL, SPECIAL,
|
||||||
|
INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE USE OR INABILITY TO USE THE
|
||||||
|
PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF DATA OR DATA BEING RENDERED INACCURATE
|
||||||
|
OR LOSSES SUSTAINED BY YOU OR THIRD PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE
|
||||||
|
WITH ANY OTHER PROGRAMS), EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE
|
||||||
|
POSSIBILITY OF SUCH DAMAGES.
|
||||||
|
|
||||||
|
### 17. Interpretation of Sections 15 and 16
|
||||||
|
|
||||||
|
If the disclaimer of warranty and limitation of liability provided above cannot be
|
||||||
|
given local legal effect according to their terms, reviewing courts shall apply local
|
||||||
|
law that most closely approximates an absolute waiver of all civil liability in
|
||||||
|
connection with the Program, unless a warranty or assumption of liability accompanies
|
||||||
|
a copy of the Program in return for a fee.
|
||||||
|
|
||||||
|
_END OF TERMS AND CONDITIONS_
|
||||||
|
|
||||||
|
## How to Apply These Terms to Your New Programs
|
||||||
|
|
||||||
|
If you develop a new program, and you want it to be of the greatest possible use to
|
||||||
|
the public, the best way to achieve this is to make it free software which everyone
|
||||||
|
can redistribute and change under these terms.
|
||||||
|
|
||||||
|
To do so, attach the following notices to the program. It is safest to attach them
|
||||||
|
to the start of each source file to most effectively state the exclusion of warranty;
|
||||||
|
and each file should have at least the “copyright” line and a pointer to
|
||||||
|
where the full notice is found.
|
||||||
|
|
||||||
|
<one line to give the program's name and a brief idea of what it does.>
|
||||||
|
Copyright (C) <year> <name of author>
|
||||||
|
|
||||||
|
This program is free software: you can redistribute it and/or modify
|
||||||
|
it under the terms of the GNU General Public License as published by
|
||||||
|
the Free Software Foundation, either version 3 of the License, or
|
||||||
|
(at your option) any later version.
|
||||||
|
|
||||||
|
This program is distributed in the hope that it will be useful,
|
||||||
|
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
||||||
|
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
||||||
|
GNU General Public License for more details.
|
||||||
|
|
||||||
|
You should have received a copy of the GNU General Public License
|
||||||
|
along with this program. If not, see <http://www.gnu.org/licenses/>.
|
||||||
|
|
||||||
|
Also add information on how to contact you by electronic and paper mail.
|
||||||
|
|
||||||
|
If the program does terminal interaction, make it output a short notice like this
|
||||||
|
when it starts in an interactive mode:
|
||||||
|
|
||||||
|
<program> Copyright (C) <year> <name of author>
|
||||||
|
This program comes with ABSOLUTELY NO WARRANTY; for details type 'show w'.
|
||||||
|
This is free software, and you are welcome to redistribute it
|
||||||
|
under certain conditions; type 'show c' for details.
|
||||||
|
|
||||||
|
The hypothetical commands `show w` and `show c` should show the appropriate parts of
|
||||||
|
the General Public License. Of course, your program's commands might be different;
|
||||||
|
for a GUI interface, you would use an “about box”.
|
||||||
|
|
||||||
|
You should also get your employer (if you work as a programmer) or school, if any, to
|
||||||
|
sign a “copyright disclaimer” for the program, if necessary. For more
|
||||||
|
information on this, and how to apply and follow the GNU GPL, see
|
||||||
|
<<http://www.gnu.org/licenses/>>.
|
||||||
|
|
||||||
|
The GNU General Public License does not permit incorporating your program into
|
||||||
|
proprietary programs. If your program is a subroutine library, you may consider it
|
||||||
|
more useful to permit linking proprietary applications with the library. If this is
|
||||||
|
what you want to do, use the GNU Lesser General Public License instead of this
|
||||||
|
License. But first, please read
|
||||||
|
<<http://www.gnu.org/philosophy/why-not-lgpl.html>>.
|
||||||
@@ -1,16 +1,20 @@
|
|||||||
# Generated by roxygen2: do not edit by hand
|
# Generated by roxygen2: do not edit by hand
|
||||||
|
|
||||||
|
export(bar_plot_fractions)
|
||||||
export(fetch_all)
|
export(fetch_all)
|
||||||
export(find_word)
|
export(find_word)
|
||||||
export(join_redner)
|
export(join_speaker)
|
||||||
|
export(party_colors)
|
||||||
export(read_all)
|
export(read_all)
|
||||||
export(read_from_csv)
|
export(read_from_csv)
|
||||||
export(repair)
|
export(repair)
|
||||||
|
export(word_usage_by_date)
|
||||||
export(write_to_csv)
|
export(write_to_csv)
|
||||||
import(dplyr)
|
import(dplyr)
|
||||||
import(pbapply)
|
import(pbapply)
|
||||||
import(purrr)
|
import(purrr)
|
||||||
import(stringr)
|
import(stringr)
|
||||||
import(tibble)
|
import(tibble)
|
||||||
|
import(tidyr)
|
||||||
import(utils)
|
import(utils)
|
||||||
import(xml2)
|
import(xml2)
|
||||||
|
|||||||
+168
-15
@@ -1,31 +1,184 @@
|
|||||||
|
#' Count number of occurences of a given word
|
||||||
|
#'
|
||||||
|
#' @param res tibble
|
||||||
|
#' @param word character
|
||||||
|
#'
|
||||||
|
#' Add number of occurences of word to talks
|
||||||
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
find_word <- function(res, word) {
|
find_word <- function(res, word) {
|
||||||
|
is_valid_res(res)
|
||||||
|
stopifnot("word must be of type character" = is.character(word))
|
||||||
talks <- res$talks
|
talks <- res$talks
|
||||||
mutate(talks, occurences = sapply(str_match_all(talks$content, regex(word, ignore_case = TRUE)),
|
mutate(
|
||||||
nrow))
|
talks,
|
||||||
|
occurences = sapply(
|
||||||
|
str_match_all(talks$content, regex(word, ignore_case = TRUE)),
|
||||||
|
nrow
|
||||||
|
)
|
||||||
|
)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#' add information from speaker table to a tibble containing speaker id
|
||||||
|
#'
|
||||||
|
#' @param tb tibble
|
||||||
|
#' @param res list of tibbles
|
||||||
|
#' @param fraction_only if TRUE, only select fraction from the resulting joined tibble
|
||||||
|
#'
|
||||||
|
#' left join speaker information from res$speaker into tb.
|
||||||
|
#' if fraction_only, drop all columns but fraction
|
||||||
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
join_redner <- function(tb, res, fraktion_only = F) {
|
join_speaker <- function(tb, res, fraction_only = F) {
|
||||||
joined <- left_join(tb, res$redner, by=c("redner" = "id"))
|
is_valid_res(res)
|
||||||
if (fraktion_only) select(joined, "fraktion")
|
stopifnot("fraction_only must be of type logical" = is.logical(fraction_only))
|
||||||
|
stopifnot("tb must be a tibble" = inherits(tb, "tbl"))
|
||||||
|
stopifnot("tb must have a speaker column" = "speaker" %in% names(tb))
|
||||||
|
|
||||||
|
joined <- left_join(tb, res$speaker, by=c("speaker" = "id"))
|
||||||
|
if (fraction_only) select(joined, "fraction")
|
||||||
else joined
|
else joined
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#' lookup table for official party colors
|
||||||
|
#'
|
||||||
|
#' @export
|
||||||
party_colors <- c(
|
party_colors <- c(
|
||||||
SPD="#DF0B25",
|
|
||||||
"CDU/CSU"="#000000",
|
|
||||||
AfD="#1A9FDD",
|
AfD="#1A9FDD",
|
||||||
"AfD&Fraktionslos"="#1A9FDD",
|
|
||||||
"DIE LINKE"="#BC3475",
|
|
||||||
"BÜNDNIS 90 / DIE GRÜNEN"="#4A932B",
|
|
||||||
FDP="#FEEB34",
|
FDP="#FEEB34",
|
||||||
Fraktionslos="#FEEB34"
|
"CDU/CSU"="#000000",
|
||||||
|
SPD="#DF0B25",
|
||||||
|
"B\u00DCNDNIS 90/DIE GR\u00DCNEN"="#4A932B",
|
||||||
|
"DIE LINKE"="#BC3475",
|
||||||
|
"AfD&Fraktionslos"="#AAAAFF",
|
||||||
|
Fraktionslos="#AAAAAA"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
party_order <- factor(c("Fraktionslos", "AfD&Fraktionslos",
|
||||||
|
"DIE LINKE", "B\u00DCNDNIS 90/DIE GR\u00DCNEN", "SPD", "CDU/CSU",
|
||||||
|
"FDP", "AfD", NA_character_))
|
||||||
|
|
||||||
|
#' Bar chart visualizing fraction based data
|
||||||
|
#'
|
||||||
|
#' Can be configured to also visualize data not related to fractions.
|
||||||
|
#'
|
||||||
|
#' @param tb tibble
|
||||||
|
#' @param x_variable column in tb, default is fraction
|
||||||
|
#' @param y_variable column in tb, default is n
|
||||||
|
#' @param fill column in tb, default is fraction
|
||||||
|
#' @param title plot title
|
||||||
|
#' @param xlab label for x axis, default is fraction
|
||||||
|
#' @param ylab label for y axis, default is n
|
||||||
|
#' @param filllab default is 'Fraction'
|
||||||
|
#' @param flipped if TRUE draw bars horizontally, else vertically. Default is TRUE
|
||||||
|
#' @param position default is 'dodge'
|
||||||
|
#' @param reorder Either reorder fraction factor by variable value or reorder fraction factor by party seat order in parliament (default).
|
||||||
|
#' @param rotatelab Default is FALSE. If true turns the labels 90 degrees to the axis.
|
||||||
|
#'
|
||||||
|
#' plot data from tb in the following way: for each item in x_variable show the corresponding value in y_variable.
|
||||||
|
#' Then color the plot depending on the fill value.
|
||||||
|
#' Give the plot a title and a label for x-axis and y-axis,
|
||||||
|
#' color the legend according to filllab and finally
|
||||||
|
#' improve positioning details according to position
|
||||||
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
bar_plot_fraktionen <- function(tb) {
|
bar_plot_fractions <- function(tb,
|
||||||
ggplot(tb, aes(x = reorder(fraktion, -n), y = n, fill = fraktion)) +
|
x_variable = NULL, # default is fraction
|
||||||
scale_fill_manual(values = party_colors) +
|
y_variable = NULL, # default is n
|
||||||
geom_bar(stat = "identity")
|
fill = NULL, # default is fraction
|
||||||
|
title = NULL,
|
||||||
|
xlab = "Fraction",
|
||||||
|
ylab = "n",
|
||||||
|
filllab = "Fraction",
|
||||||
|
flipped = TRUE,
|
||||||
|
position = "dodge",
|
||||||
|
reorder = FALSE,
|
||||||
|
rotatelab = FALSE) {
|
||||||
|
# capture expressions in arguments
|
||||||
|
fill <- enexpr(fill)
|
||||||
|
y_variable <- enexpr(y_variable)
|
||||||
|
x_variable <- enexpr(x_variable)
|
||||||
|
|
||||||
|
# set default values
|
||||||
|
if (is.null(fill)) fill <- expr(fraction)
|
||||||
|
if (is.null(y_variable)) y_variable <- expr(n)
|
||||||
|
if (is.null(x_variable)) x_variable <- expr(fraction)
|
||||||
|
|
||||||
|
# check if variables exist
|
||||||
|
if (!rlang::expr_text(x_variable) %in% names(tb))
|
||||||
|
stop(paste0(rlang::expr_text(x_variable),
|
||||||
|
" is not a column of tb. Did you set x_variable accordingly?"),
|
||||||
|
.call = NULL)
|
||||||
|
if (!rlang::expr_text(y_variable) %in% names(tb))
|
||||||
|
stop(paste0(rlang::expr_text(y_variable),
|
||||||
|
" is not a column of tb. Did you set y_variable accordingly?"),
|
||||||
|
.call = NULL)
|
||||||
|
if (!rlang::expr_text(fill) %in% names(tb))
|
||||||
|
stop(paste0(rlang::expr_text(fill),
|
||||||
|
" is not a column of tb. Did you set fill accordingly?"),
|
||||||
|
.call = NULL)
|
||||||
|
|
||||||
|
# check argument types
|
||||||
|
stopifnot("title has to be of type character or NULL" = is.character(title) || is.null(title))
|
||||||
|
stopifnot("xlab has to be of type character" = is.character(xlab))
|
||||||
|
stopifnot("ylab has to be of type character" = is.character(ylab))
|
||||||
|
stopifnot("filllab has to be of type character" = is.character(filllab))
|
||||||
|
stopifnot("flipped has to be of type logical" = is.logical(flipped))
|
||||||
|
stopifnot("rotatelab has to be of type logical" = is.logical(rotatelab))
|
||||||
|
stopifnot("reorder has to be of type logical" = is.logical(reorder))
|
||||||
|
|
||||||
|
# either reorder fraction factor by variable value
|
||||||
|
if (reorder) maps <- aes(x = reorder(!!x_variable, -!!y_variable),
|
||||||
|
y = !!y_variable,
|
||||||
|
fill = reorder(!!fill, -!!y_variable))
|
||||||
|
# or reorder fraction factor by party seat order in parliament (default)
|
||||||
|
else maps <- aes(x = factor(!!x_variable, levels = party_order),
|
||||||
|
y = !!y_variable,
|
||||||
|
fill = factor(!!fill, levels = party_order))
|
||||||
|
|
||||||
|
# make a bar plot
|
||||||
|
ggplot(tb, maps) +
|
||||||
|
scale_fill_manual(values = party_colors, na.value = "#555555") +
|
||||||
|
xlab(xlab) +
|
||||||
|
ylab(ylab) +
|
||||||
|
labs(fill = filllab) +
|
||||||
|
ggtitle(title) +
|
||||||
|
geom_bar(stat = "identity", position = position) ->
|
||||||
|
plt
|
||||||
|
|
||||||
|
# if rotatelab == TRUE, rotate x labels by 90 degrees
|
||||||
|
if (rotatelab)
|
||||||
|
plt + theme(axis.text.x = element_text(angle = 90, vjust = 0.5, hjust=1)) -> plt
|
||||||
|
|
||||||
|
# if flipped == TRUE, draw bars horizontally (default TRUE)
|
||||||
|
if (flipped) plt + coord_flip() else plt
|
||||||
|
}
|
||||||
|
|
||||||
|
#' Word usage summarised by date
|
||||||
|
#'
|
||||||
|
#' Counts how many talks do match a given pattern and summarises by date.
|
||||||
|
#'
|
||||||
|
#' @param res List of Tibbles to be analysed.
|
||||||
|
#' @param patterns Words to look up.
|
||||||
|
#' @param tidy default is FALSE.
|
||||||
|
#'
|
||||||
|
#' @export
|
||||||
|
word_usage_by_date <- function(res, patterns, tidy=F) {
|
||||||
|
is_valid_res(res)
|
||||||
|
stopifnot("patterns must be of type character" = is.character(patterns))
|
||||||
|
stopifnot("tidy must be of type logical" = is.logical(tidy))
|
||||||
|
|
||||||
|
tb <- res$talks
|
||||||
|
nms <- names(patterns)
|
||||||
|
for (i in seq_along(patterns)) {
|
||||||
|
if (!is.null(nms)) name <- nms[[i]]
|
||||||
|
else name <- patterns[[i]]
|
||||||
|
tb <- mutate(tb, {{name}} := str_count(content, patterns[[i]]))
|
||||||
|
}
|
||||||
|
left_join(tb, res$speeches, by=c("speech_id" = "id")) %>%
|
||||||
|
group_by(date) %>%
|
||||||
|
summarize(across(where(is.numeric), sum)) %>%
|
||||||
|
arrange(date) -> tb
|
||||||
|
if (!tidy) pivot_longer(tb, where(is.numeric) , names_to = "pattern", values_to="count")
|
||||||
|
else tb
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -36,12 +36,14 @@ fetch_batch <- function(offset, download_dir) {
|
|||||||
#' This fetches all available records of the 19th legislative period of the german Bundestag.
|
#' This fetches all available records of the 19th legislative period of the german Bundestag.
|
||||||
#'
|
#'
|
||||||
#' @param download_dir character
|
#' @param download_dir character
|
||||||
|
#' @param create bool
|
||||||
|
#'
|
||||||
|
#' if create is TRUE, the directory given in download_dir is created
|
||||||
#'
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
fetch_all <- function(download_dir="records/", create=FALSE) {
|
fetch_all <- function(download_dir="inst/records/", create=FALSE) {
|
||||||
# check if download_dir path is a directory path
|
# append file separator if needed
|
||||||
if (str_sub(download_dir, -1) != .Platform$file.sep)
|
download_dir <- make_directory_path(download_dir)
|
||||||
download_dir <- str_c(download_dir, .Platform$file.sep)
|
|
||||||
|
|
||||||
check_directory(download_dir, create)
|
check_directory(download_dir, create)
|
||||||
cat("Fetching all available records from bundestag.de. This may take a while ...\n")
|
cat("Fetching all available records from bundestag.de. This may take a while ...\n")
|
||||||
@@ -59,10 +61,3 @@ fetch_all <- function(download_dir="records/", create=FALSE) {
|
|||||||
# if successful, set progressbar to 100%
|
# if successful, set progressbar to 100%
|
||||||
setTimerProgressBar(pb, 250)
|
setTimerProgressBar(pb, 250)
|
||||||
}
|
}
|
||||||
|
|
||||||
stop_dir_not_creatable <- function(cond) {
|
|
||||||
# currently this has call: dir.create(download_dir)
|
|
||||||
# do we want to change this to fetch_all(...) ?
|
|
||||||
cond$message <- "Directory does not exist and can't be created. Probably because the path is not writeable."
|
|
||||||
stop(cond)
|
|
||||||
}
|
|
||||||
|
|||||||
@@ -6,6 +6,7 @@
|
|||||||
#' @import stringr
|
#' @import stringr
|
||||||
#' @import xml2
|
#' @import xml2
|
||||||
#' @import utils
|
#' @import utils
|
||||||
|
#' @import tidyr
|
||||||
#' @import purrr
|
#' @import purrr
|
||||||
#' @keywords internal
|
#' @keywords internal
|
||||||
"_PACKAGE"
|
"_PACKAGE"
|
||||||
|
|||||||
+48
@@ -18,3 +18,51 @@ check_directory <- function(path, create=F) {
|
|||||||
stop("Directory exists, but is not writeable.")
|
stop("Directory exists, but is not writeable.")
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
stop_dir_not_creatable <- function(cond) {
|
||||||
|
# currently this has call: dir.create(download_dir)
|
||||||
|
# do we want to change this to fetch_all(...) ?
|
||||||
|
cond$message <- "Directory does not exist and can't be created. Probably because the path is not writeable."
|
||||||
|
stop(cond)
|
||||||
|
}
|
||||||
|
|
||||||
|
# appends a file seperator at end of path if needed
|
||||||
|
make_directory_path <- function(path) {
|
||||||
|
if (!str_ends(path, .Platform$file.sep)) str_c(path, .Platform$file.sep)
|
||||||
|
else path
|
||||||
|
}
|
||||||
|
|
||||||
|
# check if res is of expected format
|
||||||
|
is_valid_res <- function(res) {
|
||||||
|
stopifnot("Data is missing relevant tables. Is this a return value of read_all or repair?"
|
||||||
|
= all(c("speaker", "speeches", "talks", "comments", "applause") %in% names(res)))
|
||||||
|
stopifnot("Some entries of res are no tibbles."
|
||||||
|
= all(sapply(res, typeof) == "list" & "tbl" %in% sapply(res, class)))
|
||||||
|
stopifnot("Speaker table is of wrong format."
|
||||||
|
= all(c("id", "prename", "lastname", "fraction", "title", "role_short", "role_long")
|
||||||
|
%in% names(res$speaker)) &&
|
||||||
|
all(sapply(res$speaker, is.character)))
|
||||||
|
stopifnot("Speeches table is of wrong format."
|
||||||
|
= all(c("id", "speaker", "date") %in% names(res$speeches)) &&
|
||||||
|
is.character(res$speeches$id) &&
|
||||||
|
is.character(res$speeches$speaker) &&
|
||||||
|
lubridate::is.Date(res$speeches$date))
|
||||||
|
stopifnot("Talks table is of wrong format."
|
||||||
|
= all(c("speech_id", "speaker", "content") %in% names(res$talks)) &&
|
||||||
|
all(sapply(res$talks, is.character)))
|
||||||
|
stopifnot("Comments table is of wrong format."
|
||||||
|
= all(c("speech_id", "on_speaker", "fraction", "commenter", "content")
|
||||||
|
%in% names(res$comments)) &&
|
||||||
|
all(sapply(res$comments, is.character)))
|
||||||
|
stopifnot("Applause table is of wrong format."
|
||||||
|
= all(c("speech_id", "on_speaker", "CDU_CSU", "SPD", "FDP", "DIE_LINKE", "BUENDNIS_90_DIE_GRUENEN", "AfD")
|
||||||
|
%in% names(res$applause)) &&
|
||||||
|
is.character(res$applause$speech_id) &&
|
||||||
|
is.character(res$applause$on_speaker) &&
|
||||||
|
is.logical(res$applause$`CDU_CSU`) &&
|
||||||
|
is.logical(res$applause$`SPD`) &&
|
||||||
|
is.logical(res$applause$`FDP`) &&
|
||||||
|
is.logical(res$applause$`DIE_LINKE`) &&
|
||||||
|
is.logical(res$applause$`AfD`) &&
|
||||||
|
is.logical(res$applause$`BUENDNIS_90_DIE_GRUENEN`))
|
||||||
|
}
|
||||||
|
|||||||
@@ -1,27 +1,40 @@
|
|||||||
# for usage see the example at the end
|
|
||||||
|
|
||||||
#' Parse xml records
|
#' Parse xml records
|
||||||
#'
|
#'
|
||||||
#' Creates a list of tibbles containing relevant information from all records
|
#' Creates a list of tibbles containing relevant information from all records
|
||||||
#' stored in the input directory.
|
#' stored in the input directory.
|
||||||
#'
|
#'
|
||||||
#' @param path character
|
#' @param path path to records directory
|
||||||
|
#' @param pattern search pattern to find records in directory
|
||||||
#'
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
read_all <- function(path="records/") {
|
read_all <- function(path="inst/records/", pattern="-data\\.xml") {
|
||||||
|
# append file separator if needed
|
||||||
|
path <- make_directory_path(path)
|
||||||
|
|
||||||
cat("Reading all records from", path, "\n")
|
cat("Reading all records from", path, "\n")
|
||||||
available_protocols <- list.files(path)
|
|
||||||
res <- pblapply(available_protocols, read_one, path=path)
|
|
||||||
|
|
||||||
lapply(res, `[[`, "redner") %>%
|
# list all files in directory and filter by search pattern
|
||||||
|
fs <- list.files(path)
|
||||||
|
available_protocols <- fs[str_detect(fs, pattern)]
|
||||||
|
if (length(available_protocols) == 0)
|
||||||
|
stop(paste0("The given directory does not exist or does not contain files matching \"",
|
||||||
|
pattern,
|
||||||
|
"\"."))
|
||||||
|
|
||||||
|
# parse records one by one and remove null entries
|
||||||
|
res <- compact %$% pblapply(available_protocols, read_one, path=path)
|
||||||
|
if (length(res) == 0) stop("No valid records found. Did you fetch successfully?")
|
||||||
|
|
||||||
|
lapply(res, `[[`, "speaker") %>%
|
||||||
bind_rows() %>%
|
bind_rows() %>%
|
||||||
distinct() ->
|
distinct() ->
|
||||||
redner
|
speaker
|
||||||
|
|
||||||
lapply(res, `[[`, "reden") %>%
|
lapply(res, `[[`, "speeches") %>%
|
||||||
bind_rows() %>%
|
bind_rows() %>%
|
||||||
distinct() ->
|
distinct() %>%
|
||||||
reden
|
mutate(date = as.Date(date, format="%d.%m.%Y")) ->
|
||||||
|
speeches
|
||||||
|
|
||||||
lapply(res, `[[`, "talks") %>%
|
lapply(res, `[[`, "talks") %>%
|
||||||
bind_rows() %>%
|
bind_rows() %>%
|
||||||
@@ -31,33 +44,55 @@ read_all <- function(path="records/") {
|
|||||||
lapply(res, `[[`, "comments") %>%
|
lapply(res, `[[`, "comments") %>%
|
||||||
bind_rows() %>%
|
bind_rows() %>%
|
||||||
distinct() ->
|
distinct() ->
|
||||||
comments
|
commentsandapplause
|
||||||
|
|
||||||
if (length(available_protocols) == 0)
|
filter(commentsandapplause, type == "comment") %>%
|
||||||
warning("The given directory is empty or does not exist.")
|
select(-type) ->
|
||||||
list(redner = redner, reden = reden, talks = talks, comments = comments)
|
comments
|
||||||
|
filter(commentsandapplause, type == "applause") %>%
|
||||||
|
select(-type, -commenter, -content) %>%
|
||||||
|
mutate("CDU_CSU" = str_detect(fraction, "CDU/CSU"),
|
||||||
|
"SPD" = str_detect(fraction, "SPD"),
|
||||||
|
"FDP" = str_detect(fraction, "FDP"),
|
||||||
|
"DIE_LINKE" = str_detect(fraction, "DIE LINKE"),
|
||||||
|
"BUENDNIS_90_DIE_GRUENEN" = str_detect(fraction, "B\u00DCNDNIS 90/DIE GR\u00DCNEN"),
|
||||||
|
"AfD" = str_detect(fraction, "AfD")) %>%
|
||||||
|
select(-fraction) ->
|
||||||
|
applause
|
||||||
|
|
||||||
|
list(speaker = speaker, speeches = speeches, talks = talks, comments = comments, applause = applause)
|
||||||
}
|
}
|
||||||
|
|
||||||
# this reads all currently parseable data from one xml
|
# this reads all currently parseable data from one xml
|
||||||
read_one <- function(name, path) {
|
read_one <- function(name, path) {
|
||||||
x <- tryCatch(read_xml(paste0(path, name)),
|
x <- tryCatch(read_xml(paste0(path, name)),
|
||||||
error = function(c) NULL)
|
error = function(c) NULL,
|
||||||
|
warning = function(c) NULL)
|
||||||
if (is.null(x)) return(NULL)
|
if (is.null(x)) return(NULL)
|
||||||
|
# extract date of session
|
||||||
|
date <- xml_attr(x, "sitzung-datum")
|
||||||
cs <- xml_children(x)
|
cs <- xml_children(x)
|
||||||
|
|
||||||
verlauf <- xml_find_first(x, "sitzungsverlauf")
|
verlauf <- xml_find_first(x, "sitzungsverlauf")
|
||||||
rednerl <- xml_find_first(x, "rednerliste")
|
speakerl <- xml_find_first(x, "rednerliste")
|
||||||
|
|
||||||
xml_children(rednerl) %>%
|
# check if record is invalid or empty (every record should have at least
|
||||||
parse_rednerliste() ->
|
# one speech, a speaker and a date
|
||||||
redner
|
if (is.na(date) || length(verlauf) == 0 || length(speakerl) == 0) {
|
||||||
|
warning("Invalid record found. Skipping.")
|
||||||
|
return(NULL)
|
||||||
|
}
|
||||||
|
|
||||||
|
xml_children(speakerl) %>%
|
||||||
|
parse_speakerlist() ->
|
||||||
|
speaker
|
||||||
|
|
||||||
xml_children(verlauf) %>%
|
xml_children(verlauf) %>%
|
||||||
xml_find_all("rede") %>%
|
xml_find_all("rede") %>%
|
||||||
parse_redenliste() ->
|
parse_speechlist(date) ->
|
||||||
res
|
res
|
||||||
|
|
||||||
list(redner = redner, reden = res$reden, talks = res$talks, comments = res$comments)
|
list(speaker = speaker, speeches = res$speeches, talks = res$talks, comments = res$comments)
|
||||||
}
|
}
|
||||||
|
|
||||||
xml_get <- function(node, name) {
|
xml_get <- function(node, name) {
|
||||||
@@ -66,158 +101,173 @@ xml_get <- function(node, name) {
|
|||||||
else res
|
else res
|
||||||
}
|
}
|
||||||
|
|
||||||
# parse one redner
|
# parse one speaker
|
||||||
parse_redner <- function(redner_xml) {
|
parse_speaker <- function(speaker_xml) {
|
||||||
redner_id <- xml_attr(redner_xml, "id")
|
speaker_id <- xml_attr(speaker_xml, "id")
|
||||||
nm <- xml_child(redner_xml)
|
nm <- xml_child(speaker_xml)
|
||||||
vorname <- xml_get(nm, "vorname")
|
prename <- xml_get(nm, "vorname")
|
||||||
nachname <- xml_get(nm, "nachname")
|
lastname <- xml_get(nm, "nachname")
|
||||||
fraktion <- xml_get(nm, "fraktion")
|
fraction <- xml_get(nm, "fraktion")
|
||||||
titel <- xml_get(nm, "titel")
|
title <- xml_get(nm, "titel")
|
||||||
rolle <- xml_find_all(nm, "rolle")
|
role <- xml_find_all(nm, "rolle")
|
||||||
if (length(rolle) > 0) {
|
if (length(role) > 0) {
|
||||||
rolle_lang <- xml_get(rolle, "rolle_lang")
|
role_long <- xml_get(role, "rolle_lang")
|
||||||
rolle_kurz <- xml_get(rolle, "rolle_kurz")
|
role_short <- xml_get(role, "rolle_kurz")
|
||||||
} else rolle_kurz <- rolle_lang <- NA_character_
|
} else role_short <- role_long <- NA_character_
|
||||||
c(id = redner_id, vorname = vorname, nachname = nachname, fraktion = fraktion, titel = titel,
|
c(id = speaker_id, prename = prename, lastname = lastname, fraction = fraction, title = title,
|
||||||
rolle_kurz = rolle_kurz, rolle_lang = rolle_lang)
|
role_short = role_short, role_long = role_long)
|
||||||
}
|
}
|
||||||
|
|
||||||
# parse one rede
|
# parse one speech
|
||||||
# returns: - a rede (with rede id and redner id)
|
# returns: - a speech (with speech id and speaker id)
|
||||||
# - all talks appearing in the rede (with corresponding content)
|
# - all talks appearing in the speech (with corresponding content)
|
||||||
parse_rede <- function(rede_xml) {
|
parse_speech <- function(speech_xml, date) {
|
||||||
rede_id <- xml_attr(rede_xml, "id")
|
speech_id <- xml_attr(speech_xml, "id")
|
||||||
cs <- xml_children(rede_xml)
|
cs <- xml_children(speech_xml)
|
||||||
cur_redner <- NA_character_
|
cur_speaker <- NA_character_
|
||||||
principal_redner <- NA_character_
|
principal_speaker <- NA_character_
|
||||||
cur_content <- ""
|
cur_content <- ""
|
||||||
reden <- list()
|
speeches <- list()
|
||||||
comments <- list()
|
comments <- list()
|
||||||
for (node in cs) {
|
for (node in cs) {
|
||||||
if (xml_name(node) == "p" || xml_name(node) == "name") {
|
if (xml_name(node) == "p" || xml_name(node) == "name") {
|
||||||
klasse <- xml_attr(node, "klasse")
|
klasse <- xml_attr(node, "klasse")
|
||||||
if ((!is.na(klasse) && klasse == "redner") || xml_name(node) == "name") {
|
if ((!is.na(klasse) && klasse == "redner") || xml_name(node) == "name") {
|
||||||
if (!is.na(cur_redner)) {
|
if (!is.na(cur_speaker)) {
|
||||||
rede <- c(rede_id = rede_id,
|
speech <- c(speech_id = speech_id,
|
||||||
redner = cur_redner,
|
speaker = cur_speaker,
|
||||||
content = cur_content)
|
content = cur_content)
|
||||||
reden <- c(reden, list(rede))
|
speeches <- c(speeches, list(speech))
|
||||||
cur_content <- ""
|
cur_content <- ""
|
||||||
}
|
}
|
||||||
if (is.na(principal_redner) && xml_name(node) != "name") {
|
if (is.na(principal_speaker) && xml_name(node) != "name") {
|
||||||
principal_redner <- xml_child(node) %>% xml_attr("id")
|
principal_speaker <- xml_child(node) %>% xml_attr("id")
|
||||||
}
|
}
|
||||||
if (xml_name(node) == "name") {
|
if (xml_name(node) == "name") {
|
||||||
cur_redner <- "BTP"
|
cur_speaker <- "BTP"
|
||||||
} else {
|
} else {
|
||||||
cur_redner <- xml_child(node) %>% xml_attr("id")
|
cur_speaker <- xml_child(node) %>% xml_attr("id")
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
cur_content <- paste0(cur_content, xml_text(node), sep="\n")
|
cur_content <- paste0(cur_content, xml_text(node), sep="\n")
|
||||||
}
|
}
|
||||||
} else if (xml_name(node) == "kommentar") {
|
} else if (xml_name(node) == "kommentar") {
|
||||||
# comments are of the form
|
# comments are of the form
|
||||||
# <kommentar>(blabla [Fraktion] – blabla liasdf – bla)</kommentar>
|
# <kommentar>(blabla [Fraktion] \u2013 blabla liasdf \u2013 bla)</kommentar>
|
||||||
xml_text(node) %>%
|
xml_text(node) %>%
|
||||||
str_sub(2, -2) %>%
|
str_sub(2, -2) %>%
|
||||||
str_split("–") %>%
|
str_split("\u2013") %>%
|
||||||
`[[`(1) %>%
|
`[[`(1) %>%
|
||||||
lapply(parse_comment, rede_id = rede_id, on_redner = cur_redner) ->
|
lapply(parse_comment, speech_id = speech_id, on_speaker = cur_speaker) ->
|
||||||
cs
|
cs
|
||||||
comments <- c(comments, cs)
|
comments <- c(comments, cs)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
rede <- c(rede_id = rede_id,
|
speech <- c(speech_id = speech_id,
|
||||||
redner = cur_redner,
|
speaker = cur_speaker,
|
||||||
content = cur_content)
|
content = cur_content)
|
||||||
reden <- c(reden, list(rede))
|
speeches <- c(speeches, list(speech))
|
||||||
list(rede = c(id = rede_id, redner = principal_redner),
|
list(speech = c(id = speech_id, speaker = principal_speaker, date = date),
|
||||||
parts = reden,
|
parts = speeches,
|
||||||
comments = comments)
|
comments = comments)
|
||||||
}
|
}
|
||||||
|
|
||||||
fraktionspattern <- "BÜNDNIS(SES)?\\W*90/DIE\\W*GRÜNEN|CDU/CSU|AfD|SPD|DIE LINKE|FDP|LINKEN"
|
fractionpattern <- "B\u00DCNDNIS(SES)?\\W*90/DIE\\W*GR\u00DCNEN|CDU/CSU|AfD|SPD|DIE LINKE|FDP|LINKEN"
|
||||||
fraktionsnames <- c("BÜNDNIS 90/DIE GRÜNEN", "CDU/CSU", "AfD", "SPD", "DIE LINKE", "FDP")
|
fractionnames <- c("B\u00DCNDNIS 90/DIE GR\u00DCNEN", "CDU/CSU", "AfD", "SPD", "DIE LINKE", "FDP",
|
||||||
|
"Fraktionslos")
|
||||||
|
|
||||||
parse_comment <- function(comment, rede_id, on_redner) {
|
parse_comment <- function(comment, speech_id, on_speaker) {
|
||||||
base <- c(rede_id = rede_id, on_redner = on_redner)
|
base <- c(speech_id = speech_id, on_speaker = on_speaker)
|
||||||
str_extract_all(comment, fraktionspattern) %>%
|
# classify comment
|
||||||
|
if(str_detect(comment, "Beifall")) {
|
||||||
|
str_extract_all(comment, fractionpattern) %>%
|
||||||
`[[`(1) %>%
|
`[[`(1) %>%
|
||||||
sapply(partial(flip(head), 1) %.% agrep, x=fraktionsnames, max=0.2, value=T) %>%
|
sapply(partial(flip(head), 1) %.% agrep, x=fractionnames, max=0.2, value=T) %>%
|
||||||
str_c(collapse=",") ->
|
str_c(collapse=",") ->
|
||||||
by
|
by
|
||||||
# classify comment
|
c(base, type = "applause", fraction = by, commenter = NA_character_, content = comment)
|
||||||
# TODO:
|
|
||||||
# - actually separate content properly
|
|
||||||
# - differentiate between [AfD] and AfD in by
|
|
||||||
if(str_detect(comment, "Beifall")) {
|
|
||||||
c(base, type = "applause", fraktion = by, kommentator = NA_character_, content = comment)
|
|
||||||
} else {
|
} else {
|
||||||
ps <- str_match(comment, "(.*) \\[(.*?)\\]: (.*)")[1,]
|
ps <- str_match(comment, "(.*) \\[(.*?)\\]: (.*)")[1,]
|
||||||
c(base, type = "comment", fraktion = ps[3], kommentator = ps[2], content = ps[4])
|
fraction <- agrep(ps[3], fractionnames, max=0.2, value=T)
|
||||||
|
if (all(is.na(fraction)) || length(fraction) == 0) fraction <- NA_character_
|
||||||
|
c(base, type = "comment", fraction = fraction, commenter = ps[2], content = ps[4])
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
# creates a tibble of reden and a tibble of talks from a list of xml nodes representing reden
|
# creates a tibble of speeches and a tibble of talks from a list of xml nodes representing speeches
|
||||||
parse_redenliste <- function(redenliste_xml) {
|
parse_speechlist <- function(speechlist_xml, date) {
|
||||||
d <- sapply(redenliste_xml, parse_rede)
|
d <- sapply(speechlist_xml, parse_speech, date = date)
|
||||||
reden <- simplify2array(d["rede", ])
|
speeches <- simplify2array(d["speech", ])
|
||||||
parts <- simplify2array %$% unlist(d["parts", ], recursive=FALSE)
|
parts <- simplify2array %$% unlist(d["parts", ], recursive=FALSE)
|
||||||
comments <- simplify2array %$% unlist(d["comments", ], recursive=FALSE)
|
comments <- simplify2array %$% unlist(d["comments", ], recursive=FALSE)
|
||||||
list(reden = tibble(id = reden["id",], redner = reden["redner",]),
|
list(speeches = tibble(id = speeches["id",], speaker = speeches["speaker",],
|
||||||
talks = tibble(rede_id = parts["rede_id", ],
|
date = speeches["date",]),
|
||||||
redner = parts["redner", ],
|
talks = tibble(speech_id = parts["speech_id", ],
|
||||||
|
speaker = parts["speaker", ],
|
||||||
content = parts["content", ]),
|
content = parts["content", ]),
|
||||||
comments = tibble(rede_id = comments["rede_id",],
|
comments = tibble(speech_id = comments["speech_id",],
|
||||||
on_redner = comments["on_redner",],
|
on_speaker = comments["on_speaker",],
|
||||||
type = comments["type",],
|
type = comments["type",],
|
||||||
fraktion = comments["fraktion",],
|
fraction = comments["fraction",],
|
||||||
kommentator = comments["kommentator",],
|
commenter = comments["commenter",],
|
||||||
content = comments["content", ]))
|
content = comments["content", ]))
|
||||||
}
|
}
|
||||||
|
|
||||||
# create a tibble of redner from a list of xml nodes representing redner
|
# create a tibble of speaker from a list of xml nodes representing speaker
|
||||||
parse_rednerliste <- function(rednerliste_xml) {
|
parse_speakerlist <- function(speakerliste_xml) {
|
||||||
d <- sapply(rednerliste_xml, parse_redner)
|
d <- sapply(speakerliste_xml, parse_speaker)
|
||||||
tibble(id = d["id",],
|
tibble(id = d["id",],
|
||||||
vorname = d["vorname",],
|
prename = d["prename",],
|
||||||
nachname = d["nachname",],
|
lastname = d["lastname",],
|
||||||
fraktion = d["fraktion",],
|
fraction = d["fraction",],
|
||||||
titel = d["titel",],
|
title = d["title",],
|
||||||
rolle_kurz = d["rolle_kurz",],
|
role_short = d["role_short",],
|
||||||
rolle_lang = d["rolle_lang",])
|
role_long = d["role_long",])
|
||||||
}
|
}
|
||||||
|
|
||||||
|
#' Write the parsed and repaired results into separate csv files
|
||||||
|
#'
|
||||||
|
#' @param tables list of tables to convert into a csv files.
|
||||||
|
#' @param path where to put the csv files.
|
||||||
|
#' @param create set TRUE if the path does not exist yet and you want to create it
|
||||||
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
write_to_csv <- function(tables, path="csv/", create=F) {
|
write_to_csv <- function(tables, path="inst/csv/", create=F) {
|
||||||
|
is_valid_res(tables)
|
||||||
|
stopifnot("path must be of type character" = is.character(path))
|
||||||
|
stopifnot("create must be of type logical" = is.logical(create))
|
||||||
|
|
||||||
|
path <- make_directory_path(path)
|
||||||
check_directory(path, create)
|
check_directory(path, create)
|
||||||
write.table(tables$redner, str_c(path, "redner.csv"))
|
write.table(tables$speaker, str_c(path, "speaker.csv"))
|
||||||
write.table(tables$reden, str_c(path, "reden.csv"))
|
write.table(tables$speeches, str_c(path, "speeches.csv"))
|
||||||
write.table(tables$talks, str_c(path, "talks.csv"))
|
write.table(tables$talks, str_c(path, "talks.csv"))
|
||||||
write.table(tables$comments, str_c(path, "comments.csv"))
|
write.table(tables$comments, str_c(path, "comments.csv"))
|
||||||
|
write.table(tables$applause, str_c(path, "applause.csv"))
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
#' create a tibble from the csv file
|
||||||
|
#'
|
||||||
|
#' @param path directory to read files from
|
||||||
|
#'
|
||||||
|
#' reading the tables from a csv is way faster than reading and repairing the data every single time
|
||||||
|
#'
|
||||||
#' @export
|
#' @export
|
||||||
read_from_csv <- function(path="csv/") {
|
read_from_csv <- function(path="inst/csv/") {
|
||||||
list(redner = read.table(str_c(path, "redner.csv")) %>%
|
stopifnot("path must be of type character" = is.character(path))
|
||||||
|
|
||||||
|
path <- make_directory_path(path)
|
||||||
|
list(speaker = read.table(str_c(path, "speaker.csv")) %>%
|
||||||
tibble() %>%
|
tibble() %>%
|
||||||
mutate(id = as.character(id)),
|
mutate(id = as.character(id)),
|
||||||
reden = read.table(str_c(path, "reden.csv")) %>%
|
speeches = read.table(str_c(path, "speeches.csv")) %>%
|
||||||
tibble() %>%
|
tibble() %>%
|
||||||
mutate(redner = as.character(redner)),
|
mutate(speaker = as.character(speaker),
|
||||||
|
date = as.Date(date)),
|
||||||
talks = tibble %$% read.table(str_c(path, "talks.csv")),
|
talks = tibble %$% read.table(str_c(path, "talks.csv")),
|
||||||
comments = tibble %$% read.table(str_c(path, "comments.csv")))
|
comments = tibble %$% read.table(str_c(path, "comments.csv")),
|
||||||
|
applause = tibble %$% read.table(str_c(path, "applause.csv"))) -> res
|
||||||
|
is_valid_res(res)
|
||||||
|
res
|
||||||
}
|
}
|
||||||
|
|
||||||
# -------------------------------
|
|
||||||
# EXAMPLE USE
|
|
||||||
|
|
||||||
# make sure data ist downloaded via fetch.R
|
|
||||||
# res <- read_one("records/19126-data.xml")
|
|
||||||
#
|
|
||||||
# res$redner
|
|
||||||
# res$reden
|
|
||||||
# res$talks
|
|
||||||
|
|
||||||
# -------------------------------
|
|
||||||
|
|||||||
+71
-47
@@ -1,15 +1,16 @@
|
|||||||
fraktionen <- c("AFD" = "AfD",
|
fractions <- c("AFD" = "AfD",
|
||||||
"BÜNDNIS90/" = "BÜNDNIS 90 / DIE GRÜNEN",
|
"AFD&FRAKTIONSLOS" = "AfD&Fraktionslos",
|
||||||
"BÜNDNIS90/DIEGRÜNEN" = "BÜNDNIS 90 / DIE GRÜNEN",
|
"B\u00DCNDNIS90/" = "B\u00DCNDNIS 90/DIE GR\u00DCNEN",
|
||||||
|
"B\u00DCNDNIS90/DIEGR\u00DCNEN" = "B\u00DCNDNIS 90/DIE GR\u00DCNEN",
|
||||||
"FRAKTIONSLOS" = "Fraktionslos",
|
"FRAKTIONSLOS" = "Fraktionslos",
|
||||||
"DIELINKE" = "DIE LINKE",
|
"DIELINKE" = "DIE LINKE",
|
||||||
"SPD" = "SPD",
|
"SPD" = "SPD",
|
||||||
"CDU/CSU" = "CDU/CSU",
|
"CDU/CSU" = "CDU/CSU",
|
||||||
"FDP" = "FDP")
|
"FDP" = "FDP")
|
||||||
|
|
||||||
repair_fraktion <- function(fraktion) {
|
repair_fraction <- function(fraction) {
|
||||||
cleaned <- str_to_upper %$% str_replace_all(fraktion, "\\s", "")
|
cleaned <- str_to_upper %$% str_replace_all(fraction, "\\s", "")
|
||||||
fraktionen[cleaned]
|
fractions[cleaned]
|
||||||
}
|
}
|
||||||
|
|
||||||
# takes vector of titel and keeps longest
|
# takes vector of titel and keeps longest
|
||||||
@@ -21,44 +22,51 @@ longest_titel <- function(titel) {
|
|||||||
# takes character vector, removes duplicates and collapses
|
# takes character vector, removes duplicates and collapses
|
||||||
collect_unique <- function(xs) xs %>% clear_na() %>% unique() %>% str_c(collapse="&") %>% na_if("")
|
collect_unique <- function(xs) xs %>% clear_na() %>% unique() %>% str_c(collapse="&") %>% na_if("")
|
||||||
|
|
||||||
# expects a tibble of redner and repairs
|
# expects a tibble of speaker and repairs
|
||||||
repair_redner <- function(redner) {
|
repair_speaker <- function(speaker) {
|
||||||
if (nrow(redner) == 0) return(redner)
|
if (nrow(speaker) == 0) return(speaker)
|
||||||
redner %>%
|
speaker %>%
|
||||||
filter(id != "10000") %>% # invalid id's
|
filter(id != "10000") %>% # invalid id's
|
||||||
mutate(fraktion = Vectorize(repair_fraktion)(fraktion)) %>% # fix fraktion
|
mutate(fraction = Vectorize(repair_fraction)(fraction)) %>% # fix fraction
|
||||||
group_by(id) %>%
|
group_by(id) %>%
|
||||||
summarize(vorname = head(vorname, 1),
|
summarize(prename = head(prename, 1),
|
||||||
nachname = head(nachname, 1),
|
lastname = head(lastname, 1),
|
||||||
fraktion = collect_unique(fraktion),
|
fraction = collect_unique(fraction),
|
||||||
titel = longest_titel(titel),
|
title = longest_titel(title),
|
||||||
rolle_kurz = collect_unique(str_squish(rolle_kurz)),
|
role_short = collect_unique(str_squish(role_short)),
|
||||||
rolle_lang = collect_unique(str_squish(rolle_lang))) %>%
|
role_long = collect_unique(str_squish(role_long))) %>%
|
||||||
ungroup() #%>%
|
ungroup() #%>%
|
||||||
# arrange(id) %>%
|
|
||||||
# distinct(vorname, nachname, fraktion, titel)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
repair_reden <- function(reden) {
|
repair_speeches <- function(speeches) {
|
||||||
if (nrow(reden) == 0) return(reden)
|
if (nrow(speeches) == 0) return(speeches)
|
||||||
# TODO: fill with content
|
# TODO: fill with content
|
||||||
reden
|
speeches
|
||||||
}
|
}
|
||||||
|
|
||||||
repair_talks <- function(talks) {
|
repair_talks <- function(talks) {
|
||||||
if (nrow(talks) == 0) return(talks)
|
if (nrow(talks) == 0) return(talks)
|
||||||
# TODO: fill with content
|
# ignore all talks which have empty content
|
||||||
talks
|
filter(talks, str_length(content) > 0)
|
||||||
}
|
}
|
||||||
|
|
||||||
# tries to find the correct redner id given a name
|
#' Lookup name in speakers table
|
||||||
# this is sufficient since every prename lastname combination in the bundestag is
|
#'
|
||||||
# unique (luckily :D)
|
#' Tries to find the correct speaker id given a name.
|
||||||
# returns a lookup table
|
#' This is sufficient since every prename lastname combination in the bundestag is
|
||||||
lookup_redner <- function(comments, redner) {
|
#' unique (luckily :D)
|
||||||
tobereplaced <- "[-–—‑- ]"
|
#'
|
||||||
redner %>%
|
#' @param tb tibble
|
||||||
unite(name, vorname, nachname, sep=".*") %>%
|
#' @param speaker tibble
|
||||||
|
#' @param name_variable name
|
||||||
|
#'
|
||||||
|
#' Tries to match the name_variable column with speaker names
|
||||||
|
#'
|
||||||
|
#' returns a lookup table
|
||||||
|
lookup_speaker <- function(tb, speaker, name_variable) {
|
||||||
|
tobereplaced <- "[\u002D\u2013\u2014\u2011\u00AD ]"
|
||||||
|
speaker %>%
|
||||||
|
unite(name, prename, lastname, sep=".*") %>%
|
||||||
mutate(name = str_replace_all(name, tobereplaced, ".*")) ->
|
mutate(name = str_replace_all(name, tobereplaced, ".*")) ->
|
||||||
rs
|
rs
|
||||||
find_match <- function(komm) {
|
find_match <- function(komm) {
|
||||||
@@ -68,28 +76,44 @@ lookup_redner <- function(comments, redner) {
|
|||||||
if (length(matches) == 0) return(NA_character_)
|
if (length(matches) == 0) return(NA_character_)
|
||||||
rs[head(matches, 1), ]$id
|
rs[head(matches, 1), ]$id
|
||||||
}
|
}
|
||||||
comments %>%
|
tb %>%
|
||||||
distinct(kommentator) %>%
|
distinct({{name_variable}}) %>%
|
||||||
mutate(redner = Vectorize(find_match)(str_replace_all(kommentator, tobereplaced, "")))
|
mutate(speaker = Vectorize(find_match)(str_replace_all({{name_variable}}, tobereplaced, "")))
|
||||||
}
|
}
|
||||||
|
|
||||||
repair_comments <- function(comments, redner) {
|
repair_comments <- function(comments, speaker, lookup_speaker=F) {
|
||||||
# try to find a redner id for each actual comment
|
|
||||||
comments %>%
|
comments %>%
|
||||||
filter(!is.na(kommentator)) %>%
|
filter(!is.na(commenter) | !is.na(content) | !is.na(fraction)) ->
|
||||||
lookup_redner(redner) %>%
|
tb
|
||||||
left_join(comments, ., by="kommentator") %>%
|
if (lookup_speaker) {
|
||||||
select(-kommentator)
|
cat(paste0("Looking up speaker id's for names in comments. This may take a while ...\n",
|
||||||
|
"Use repair(, lookup_speaker = FALSE) to skip this.\n"))
|
||||||
|
# try to find a speaker id for each actual comment
|
||||||
|
tb %>%
|
||||||
|
filter(!is.na(commenter)) %>%
|
||||||
|
lookup_speaker(speaker, commenter) %>%
|
||||||
|
left_join(tb, ., by="commenter")
|
||||||
|
} else tb
|
||||||
}
|
}
|
||||||
|
|
||||||
#' Repair parsed tables
|
#' Repair parsed tables
|
||||||
#'
|
#'
|
||||||
|
#' @param parse_output tibble
|
||||||
|
#' @param lookup_speaker bool
|
||||||
|
#'
|
||||||
|
#' If lookup_speaker is TRUE, members of the parliament mentioned in comments are looked up in speaker table.
|
||||||
|
#'
|
||||||
|
#' Possible test: check identical(repair(res), repair(repair(res))) == TRUE
|
||||||
|
#' Since repaired tables should be a fixpoint of repair.
|
||||||
#' @export
|
#' @export
|
||||||
repair <- function(parse_output) {
|
repair <- function(parse_output, lookup_speaker = FALSE) {
|
||||||
list(redner = repair_redner(parse_output$redner),
|
is_valid_res(parse_output)
|
||||||
reden = repair_reden(parse_output$reden),
|
stopifnot("lookup_speaker must be of type logical" = is.logical(lookup_speaker))
|
||||||
|
list(speaker = repair_speaker(parse_output$speaker),
|
||||||
|
speeches = repair_speeches(parse_output$speeches),
|
||||||
talks = repair_talks(parse_output$talks),
|
talks = repair_talks(parse_output$talks),
|
||||||
#comments = repair_comments(parse_output$comments)
|
comments = repair_comments(parse_output$comments,
|
||||||
comments = parse_output$comments
|
parse_output$speaker,
|
||||||
)
|
lookup_speaker),
|
||||||
|
applause = parse_output$applause)
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -1,90 +1,178 @@
|
|||||||
# How to develop
|
# Description
|
||||||
|
|
||||||
Wie kann man entwickeln?
|
R package to analyze parliamentary records of the 19th legislative period of the Bundestag,
|
||||||
|
the German parliament.
|
||||||
|
|
||||||
|
# Installation
|
||||||
|
|
||||||
|
Using the `remotes` package, this is easily installed via:
|
||||||
```r
|
```r
|
||||||
# alles geht mit devtools (laedt auch noch ein paar andere pakete)
|
remotes::install_url("https://git.flavigny.de/christian/hateimparlament/archive/master.zip")
|
||||||
library(devtools)
|
```
|
||||||
|
Since the fetching and reading is very slow and depends on an internet connection, all vignettes
|
||||||
|
use `read_from_csv` to read already parsed tibbles from `.csv` files.
|
||||||
|
|
||||||
# neu laden aller paket funktionen
|
That's why, if you want to build the vignettes yourself, you need to
|
||||||
|
download the source code, e.g. on Linux
|
||||||
|
```
|
||||||
|
git clone https://git.flavigny.de/christian/hateimparlament
|
||||||
|
cd hateimparlament
|
||||||
|
```
|
||||||
|
then start `R` and do
|
||||||
|
```r
|
||||||
|
devtools::load_all()
|
||||||
|
fetch_all(create = TRUE)
|
||||||
|
read_all() %>% repair() -> res
|
||||||
|
write_to_csv(res, create = TRUE)
|
||||||
|
```
|
||||||
|
Then finally, do:
|
||||||
|
```r
|
||||||
|
devtools::install(build_vignettes = TRUE)
|
||||||
|
```
|
||||||
|
|
||||||
|
# Features
|
||||||
|
|
||||||
|
The package mainly supplies 4 functionalities:
|
||||||
|
|
||||||
|
## Download records
|
||||||
|
|
||||||
|
To analyze records, they need to be downloaded. This is done with `fetch_all`:
|
||||||
|
```r
|
||||||
|
fetch_all("records/", create = TRUE) # path to directory where records should be stored
|
||||||
|
```
|
||||||
|
This downloads all parliamentary records and stores them as `.xml` files in the given directory.
|
||||||
|
|
||||||
|
## Parse records
|
||||||
|
|
||||||
|
To use the records in R, they are converted to `tibble`s with
|
||||||
|
```r
|
||||||
|
res_raw <- read_all("records/") # path to directory where records are stored
|
||||||
|
```
|
||||||
|
|
||||||
|
`res_raw` is a named list with 5 `tibble`s:
|
||||||
|
|
||||||
|
### Speaker
|
||||||
|
|
||||||
|
Table of all speakers of this legislative period.
|
||||||
|
|
||||||
|
Fields:
|
||||||
|
- `id`: Unique speaker id
|
||||||
|
- `prename`: Prename
|
||||||
|
- `lastname`: Surname
|
||||||
|
- `fraction`: Name of fraction if the speaker is member of parliament.
|
||||||
|
- `title`: Title, e.g. ,,Prof''
|
||||||
|
- `role_short`: Short name of role, e.g. ,,Bundeskanzlerin''
|
||||||
|
- `role_long`: Long name of role
|
||||||
|
|
||||||
|
### Speeches
|
||||||
|
|
||||||
|
Table of all speeches given during this legislative period.
|
||||||
|
|
||||||
|
Fields:
|
||||||
|
- `id`: Unique speech id
|
||||||
|
- `speaker`: Principal speaker (the person standing behind the lectern during the speech).
|
||||||
|
- `date`: Date of session
|
||||||
|
|
||||||
|
### Talks
|
||||||
|
|
||||||
|
Within a speech, there can be multiple talks by different people. Mostly this is the main speech
|
||||||
|
by the principal speaker, but usually there are questions by other members of parliament or
|
||||||
|
order calls by the president of the Bundestag.
|
||||||
|
|
||||||
|
Fields:
|
||||||
|
- `speech_id`: Speech in which this talk has been given
|
||||||
|
- `speaker`: Person that actually talks
|
||||||
|
- `content`: Spoken content
|
||||||
|
|
||||||
|
### Comments
|
||||||
|
|
||||||
|
These are the interjections that appear during the speeches.
|
||||||
|
|
||||||
|
Fields:
|
||||||
|
- `speech_id`: The speech that was interrupted
|
||||||
|
- `on_speaker`: The speaker who was interrupted
|
||||||
|
- `fraction`: The fraction of the commenter
|
||||||
|
- `commenter`: The person who interrupted the speech
|
||||||
|
- `comment`: The content of the comment
|
||||||
|
|
||||||
|
### Applause
|
||||||
|
|
||||||
|
Table containing all the rounds of applause that happened during this legislative period.
|
||||||
|
|
||||||
|
Fields:
|
||||||
|
- `speech_id`: Speech during which was applauded
|
||||||
|
- `on_speaker`: Speaker who was applauded
|
||||||
|
|
||||||
|
And then logical fields `CDU_CSU`, `SPD`, `FDP`, `DIE_LINKE`, `BUENDNIS_90_DIE_GRUENEN`, `AfD`
|
||||||
|
for every fraction in the Bundestag, signifying whether this fraction applauded.
|
||||||
|
|
||||||
|
## Repair records
|
||||||
|
|
||||||
|
The parliamentary records usually contain some major and minor formatting issues. These are
|
||||||
|
mostly resolved by using
|
||||||
|
```
|
||||||
|
res <- repair(res_raw)
|
||||||
|
```
|
||||||
|
By passing `lookup_speaker = TRUE`, even commenters in
|
||||||
|
`res_raw$comments` are matched with their respective speaker id.
|
||||||
|
|
||||||
|
## Analysis
|
||||||
|
|
||||||
|
Also some functions are provided to analyze the parliamentary records and draw some plots:
|
||||||
|
|
||||||
|
- `bar_plot_fractions`
|
||||||
|
- `find_word`
|
||||||
|
- `join_speaker`
|
||||||
|
- `word_usage_by_date`
|
||||||
|
|
||||||
|
See their usage with the `?` operator.
|
||||||
|
|
||||||
|
In the vignettes you can find different analyses of the protocols, for example:
|
||||||
|
|
||||||
|
- "Who talks the most?"
|
||||||
|
- "Which party gives the most speeches?"
|
||||||
|
- "Which party comments the most on which parties?"
|
||||||
|
- "When are which topics discussed the most?"
|
||||||
|
- ...
|
||||||
|
|
||||||
|
# Contributing
|
||||||
|
|
||||||
|
Developing works the easiest with `devtools`:
|
||||||
|
```r
|
||||||
|
library(devtools)
|
||||||
|
```
|
||||||
|
When you changed something or added some functionality, you can reload all package functions with
|
||||||
|
```r
|
||||||
load_all()
|
load_all()
|
||||||
```
|
```
|
||||||
Wir verwenden NIEMALS source, etc.! Außerdem NIEMALD library(...) verwenden, sondern
|
If you want to avoid reading all records every time you start a new R session, you can
|
||||||
um neue pakete hinzuzufuegen (als dependency), verwende:
|
write your parsed tibbles to CSV files:
|
||||||
|
|
||||||
|
```
|
||||||
|
tables <- read_all()
|
||||||
|
tables <- repair(tables)
|
||||||
|
write_to_csv(tables, "path/to/csv/")
|
||||||
|
```
|
||||||
|
Then later you can use
|
||||||
|
```r
|
||||||
|
res <- read_from_csv("path/to/csv/")
|
||||||
|
```
|
||||||
|
to load your stored tibbles very fast.
|
||||||
|
|
||||||
|
NEVER use source(...), etc.! Also NEVER use library(...).
|
||||||
|
To add new packages (as dependency), use:
|
||||||
```r
|
```r
|
||||||
use_package("my-good-old-package")
|
use_package("my-good-old-package")
|
||||||
```
|
```
|
||||||
Um paket imports verfuegbar zu machen, muss man diese in `R/hateimparlament-package.R`
|
To make package imports available, you have to add them to `R/hateimparlament-package.R`
|
||||||
als `@import <package>` hinzufuegen.
|
as `@import <package>`.
|
||||||
|
|
||||||
Um dokumentationen neu zu laden / zu erstellen (ruft roxgen auf)
|
To reload / create documentation (calls roxygen)
|
||||||
```r
|
```r
|
||||||
document()
|
document()
|
||||||
```
|
```
|
||||||
|
|
||||||
Baue vignetten
|
Build vignettes
|
||||||
```r
|
```r
|
||||||
rmarkdown::render("vignettes/bla.Rmd")
|
rmarkdown::render("vignettes/test.Rmd")
|
||||||
```
|
```
|
||||||
|
|
||||||
# Herunterladen
|
|
||||||
|
|
||||||
Bevor analysiert werden kann, muss fetch.R ausgeführt werden, um alle Protokolle herunterzuladen.
|
|
||||||
|
|
||||||
# Parsing
|
|
||||||
|
|
||||||
## Tabellen
|
|
||||||
|
|
||||||
parse.R parsed einzelne Protokolle und erstellt 3 Tibbles
|
|
||||||
|
|
||||||
### Redner
|
|
||||||
|
|
||||||
Struktur: `id` , `vorname` , `nachname` , `fraktion` , `titel` , `rolle_kurz`, `rolle_lang`
|
|
||||||
|
|
||||||
Die Rollen sind beispielsweise "Bundeskanzlerin". Leider gegendert und deshalb wahrscheinlich
|
|
||||||
nervig zu analysieren.
|
|
||||||
|
|
||||||
Wird gewonnnen aus dem `<rednerliste>` Eintrag am Ende der Protokolle.
|
|
||||||
|
|
||||||
### Reden
|
|
||||||
|
|
||||||
Struktur: `id` , `redner`
|
|
||||||
|
|
||||||
Die Reden `id` wird im Protokoll festgelegt und ist eindeutig. Eine Rede ist ein
|
|
||||||
`<rede>` Eintrag im Sitzungsverlauf. Eine Rede hat immer einen Hauptredner
|
|
||||||
(der der vorne am Pult steht).
|
|
||||||
|
|
||||||
Innerhalb einer Rede kann es verschieden Redebeiträge geben:
|
|
||||||
|
|
||||||
- Kommentare: Beifall, Zwischenrufe, etc.
|
|
||||||
- Redebeiträge: Typischerweise hauptsächlich der Hauptredner, aber auch Zwischenfragen. Diese werden
|
|
||||||
beim parsen in der Tabelle Talks gespeichert.
|
|
||||||
|
|
||||||
### Talks
|
|
||||||
|
|
||||||
Struktur: `rede_id` , `redner` , `content`
|
|
||||||
|
|
||||||
Das sind die eigentlichen Redebeiträge, die innerhalb von _rede_ Einträgen auftauchen. Dabei gilt:
|
|
||||||
|
|
||||||
- `rede_id`: Die Rede in dem der Beitrag auftaucht
|
|
||||||
- `redner`: Der Sprecher des Redebeitrags
|
|
||||||
- `content`: Der Inhalt der Rede (__wichtig__: Aktuell werden die Ordnungskommentare des
|
|
||||||
Bundestagspräsidenten nicht herausgefiltert, tauchen also im Inhalt auf, obwohl sie nicht vom
|
|
||||||
`redner` gesprochen werden. To be fixed -> Issues!)
|
|
||||||
|
|
||||||
## Noch zu parsen: Alles kann, nichts muss.
|
|
||||||
|
|
||||||
- Kommentare (aktuell werden nur `<p>`'s in Reden gesammelt). Hier ist zu überlegen, wie diese
|
|
||||||
gesammelt werden sollten.
|
|
||||||
- Meta Daten? Diese sind teilweise in den `rede_id`'s encoded.
|
|
||||||
|
|
||||||
## Kombinieren der Tabellen der Protokolle
|
|
||||||
|
|
||||||
- Alle Tabellen sollten schlussendlich kombiniert werden zu großen Tabellen über
|
|
||||||
alle Protokolle.
|
|
||||||
|
|
||||||
# Analyse
|
|
||||||
|
|
||||||
- Schnittmenge AfD Vokabular und Hitler's Reden?
|
|
||||||
- Redeanteile nach Geschlecht (dazu gibt es leider keine Daten in der Rednerliste), Fraktion, etc.
|
|
||||||
- Ideen, Ideen, Ideen ...
|
|
||||||
|
|||||||
@@ -1,8 +0,0 @@
|
|||||||
import os
|
|
||||||
words = []
|
|
||||||
for i in range(1, 7):
|
|
||||||
with open(f'hitler_rede_{i}') as f:
|
|
||||||
lines = f.readlines()
|
|
||||||
for line in lines:
|
|
||||||
words.extend(line.split(sep=" "))
|
|
||||||
|
|
||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,63 @@
|
|||||||
|
import re
|
||||||
|
|
||||||
|
german_words = []
|
||||||
|
with open('/home/josua/deu_mixed-typical_2011_1M/deu_mixed-typical_2011_1M-words.txt') as f:
|
||||||
|
lines = f.readlines()
|
||||||
|
for line in lines:
|
||||||
|
#print(line.split(sep="\t"))
|
||||||
|
index, word, count = line.split(sep="\t")
|
||||||
|
if int(index) > 100 and int(count) > 5:
|
||||||
|
german_words.append(word.lower())
|
||||||
|
|
||||||
|
with open('/home/josua/deu_mixed-typical_2011_1M/deu_news_1995_1M-words.txt') as f:
|
||||||
|
lines = f.readlines()
|
||||||
|
for line in lines:
|
||||||
|
#print(line.split(sep="\t"))
|
||||||
|
index, word, count = line.split(sep="\t")
|
||||||
|
if int(index) > 100 and int(count) > 5:# only words that are used more than 5 times
|
||||||
|
german_words.append(word.lower())
|
||||||
|
|
||||||
|
|
||||||
|
def get_words_from_line(line):
|
||||||
|
words = line.split(sep=" ")
|
||||||
|
ret_list = []
|
||||||
|
for word in words:
|
||||||
|
word = re.sub("[^a-zA-ZüöäÜÖÄßẞ]", "", word)
|
||||||
|
ret_list.append(word.lower())
|
||||||
|
return ret_list
|
||||||
|
|
||||||
|
|
||||||
|
hitler_words = []
|
||||||
|
for i in range(1, 7):
|
||||||
|
with open(f'hitler_rede_{i}') as f:
|
||||||
|
lines = f.readlines()
|
||||||
|
for line in lines:
|
||||||
|
hitler_words.extend(get_words_from_line(line))
|
||||||
|
|
||||||
|
with open(f'goebbels_sportpalast') as f:
|
||||||
|
lines = f.readlines()
|
||||||
|
for line in lines:
|
||||||
|
hitler_words.extend(get_words_from_line(line))
|
||||||
|
|
||||||
|
with open(f'mein_kampf') as f:
|
||||||
|
lines = f.readlines()
|
||||||
|
for line in lines:
|
||||||
|
hitler_words.extend(get_words_from_line(line))
|
||||||
|
|
||||||
|
german_words = set(german_words)
|
||||||
|
hitler_words = set(hitler_words) #unique
|
||||||
|
#filter_words = hitler_words.intersection(set(german_words))
|
||||||
|
|
||||||
|
only_hitler_words = list(hitler_words.difference(german_words))
|
||||||
|
|
||||||
|
print(only_hitler_words)
|
||||||
|
|
||||||
|
with open("german_words", "w") as f:
|
||||||
|
for word in german_words:
|
||||||
|
word += "\n"
|
||||||
|
f.write(word)
|
||||||
|
|
||||||
|
with open("hitler_words", "w") as f:
|
||||||
|
for word in only_hitler_words:
|
||||||
|
word += "\n"
|
||||||
|
f.write(word)
|
||||||
+108989
File diff suppressed because it is too large
Load Diff
Binary file not shown.
@@ -0,0 +1,16 @@
|
|||||||
|
\documentclass{article}
|
||||||
|
\usepackage[top=2.5cm, bottom=2.5cm]{geometry}
|
||||||
|
|
||||||
|
\begin{document}
|
||||||
|
\section*{Projektbeschreibung}
|
||||||
|
Wir haben zunächst die Plenarprotokolle der 19. Wahlperiode von der Website automatisiert herunterladen lassen.
|
||||||
|
Als nächstes haben wir die Daten in ein für die Analyse sinnvolles Format gebracht, d.h. 5 Tibbles und Fehler ausgebessert.
|
||||||
|
Daraufhin konnten wir mit der Analyse beginnen.
|
||||||
|
Insbesondere
|
||||||
|
\section*{Werkzeuge aus der Vorlesung}
|
||||||
|
Wir haben, da es hauptsächlich um Datenanalyse ging, sehr viel mit tidyverse gearbeitet.
|
||||||
|
Ganz zu Beginn haben wir fürs fetchen der Protokolle rvest verwendet.
|
||||||
|
Für die Visualisierung haben wir ggplot2 sowie vignettes genutzt.
|
||||||
|
\section*{Organisation des Teams}
|
||||||
|
\section*{Meine Beteiligung}
|
||||||
|
\end{document}
|
||||||
Binary file not shown.
@@ -0,0 +1,171 @@
|
|||||||
|
\documentclass{beamer}
|
||||||
|
|
||||||
|
\usepackage[utf8]{inputenc}
|
||||||
|
|
||||||
|
\usepackage{listings}
|
||||||
|
\lstdefinestyle{mystyle}{
|
||||||
|
commentstyle=\color{gray},
|
||||||
|
keywordstyle=\color{black},
|
||||||
|
numberstyle=\tiny\color{gray},
|
||||||
|
stringstyle=\color{black},
|
||||||
|
basicstyle=\ttfamily\footnotesize,
|
||||||
|
breakatwhitespace=false,
|
||||||
|
breaklines=true,
|
||||||
|
captionpos=b,
|
||||||
|
keepspaces=true,
|
||||||
|
numbers=left,
|
||||||
|
numbersep=5pt,
|
||||||
|
showspaces=false,
|
||||||
|
showstringspaces=false,
|
||||||
|
showtabs=false,
|
||||||
|
tabsize=2
|
||||||
|
}
|
||||||
|
|
||||||
|
\lstset{style=mystyle}
|
||||||
|
\begin{document}
|
||||||
|
|
||||||
|
\begin{frame}
|
||||||
|
\frametitle{Implementierung}
|
||||||
|
\tableofcontents
|
||||||
|
\end{frame}
|
||||||
|
|
||||||
|
\section{Herunterladen der Protokolle}
|
||||||
|
\begin{frame}
|
||||||
|
\frametitle{Herunterladen der Protokolle}
|
||||||
|
|
||||||
|
Funktion: \lstinline{fetch_all(download_dir)}
|
||||||
|
|
||||||
|
\begin{itemize}[<+->]
|
||||||
|
\item Protokolle als XML-Dateien von \url{bundestag.de} herunterladen und
|
||||||
|
in \lstinline{download_dir} speichern.
|
||||||
|
\item Problem: Maschinenunfreundliche Webseite
|
||||||
|
\item Lösung: Source Code von \url{bundestag.de} nach Schnittstelle durchsuchen
|
||||||
|
\end{itemize}
|
||||||
|
|
||||||
|
\end{frame}
|
||||||
|
\section{Konvertierung der XML-Dateien in tibbles}
|
||||||
|
|
||||||
|
\begin{frame}
|
||||||
|
\frametitle{Konvertierung der XML-Dateien in tibbles}
|
||||||
|
Funktion: \lstinline{read_all(filepath)}
|
||||||
|
\begin{itemize}[<+->]
|
||||||
|
\item Liest jede XML-Datei in angegebenem Dateipfad einzeln
|
||||||
|
\item Extrahiert Sitzungsdatum, Rednerliste und Sitzungsverlauf
|
||||||
|
\item Konvertiert Rednerliste in eine R Liste.
|
||||||
|
\item Iteriert durch den Sitzungsverlauf, extrahiert Reden,
|
||||||
|
Redebeiträge, Kommentare und Beifall
|
||||||
|
\item Kombiniert alle Redner, Reden, Redebeiträge, Kommentare und Beifall
|
||||||
|
zu 5 tibbles und gibt benannte Liste zurück.
|
||||||
|
\end{itemize}
|
||||||
|
\end{frame}
|
||||||
|
|
||||||
|
\begin{frame}[fragile]
|
||||||
|
\frametitle{Tabellen}
|
||||||
|
Ergebnis der Konvertierung ist eine benannte Liste \lstinline{res} mit tibbles:
|
||||||
|
\pause
|
||||||
|
\begin{lstlisting}[language=R,basicstyle=\tiny\ttfamily]
|
||||||
|
> res$speaker
|
||||||
|
# A tibble: 1,025 x 7
|
||||||
|
id prename lastname fraction title role_short role_long
|
||||||
|
<chr> <chr> <chr> <chr> <chr> <chr> <chr>
|
||||||
|
1 110021 Alterspraesident D Otto Solms NA NA Alterspraesi Alterspraesi
|
||||||
|
2 110032 Carsten Schneider SPD NA NA NA
|
||||||
|
# with 1,023 more rows
|
||||||
|
\end{lstlisting}
|
||||||
|
\pause
|
||||||
|
\begin{lstlisting}[language=R,basicstyle=\tiny\ttfamily]
|
||||||
|
> res$speeches
|
||||||
|
# A tibble: 25,068 x 3
|
||||||
|
id speaker date
|
||||||
|
<chr> <chr> <date>
|
||||||
|
1 ID19100100 11002190 2017-10-24
|
||||||
|
2 ID19100200 11002190 2017-10-24
|
||||||
|
# with 25,066 more rows
|
||||||
|
\end{lstlisting}
|
||||||
|
\pause
|
||||||
|
\begin{lstlisting}[language=R,basicstyle=\tiny\ttfamily]
|
||||||
|
> res$talks
|
||||||
|
# A tibble: 63,663 x 3
|
||||||
|
speech_id speaker content
|
||||||
|
<chr> <chr> <chr>
|
||||||
|
1 ID19100100 11002190 "Guten Morgen, liebe Kolleginnen und Kollegen! Nehmen Sie
|
||||||
|
2 ID19100300 11003218 "Sehr geehrter Herr Praesident! Sehr geehrte Kolleginnen u
|
||||||
|
# with 63,661 more rows
|
||||||
|
\end{lstlisting}
|
||||||
|
|
||||||
|
\end{frame}
|
||||||
|
|
||||||
|
\begin{frame}[fragile]
|
||||||
|
\begin{lstlisting}[language=R,basicstyle=\tiny\ttfamily]
|
||||||
|
> res$comments
|
||||||
|
# A tibble: 83,649 x 5
|
||||||
|
speech_id on_speaker fraction commenter content
|
||||||
|
<chr> <chr> <chr> <chr> <chr>
|
||||||
|
1 ID19100300 11003218 BUENDNIS 90/D Katrin Goering Was?
|
||||||
|
2 ID19100300 11003218 CDU/CSU Volker Kauder Warum habt ihr das bei Ge
|
||||||
|
# with 83,647 more rows
|
||||||
|
\end{lstlisting}
|
||||||
|
\pause
|
||||||
|
\begin{lstlisting}[language=R,basicstyle=\tiny\ttfamily]
|
||||||
|
> res$applause
|
||||||
|
# A tibble: 89,586 x 8
|
||||||
|
speech_id on_speaker CDU_CSU SPD FDP DIE_LINKE BUENDNIS_90_DIE_GRU AfD
|
||||||
|
<chr> <chr> <lgl> <lgl> <lgl> <lgl> <lgl> <lgl>
|
||||||
|
1 ID19100300 11003218 FALSE TRUE FALSE TRUE TRUE FALSE
|
||||||
|
2 ID19100300 11003218 FALSE TRUE TRUE TRUE FALSE FALSE
|
||||||
|
# with 89,584 more rows\end{lstlisting}
|
||||||
|
\end{frame}
|
||||||
|
|
||||||
|
\section{Reparieren von Fehlern}
|
||||||
|
\begin{frame}
|
||||||
|
\frametitle{Reparieren von Fehlern}
|
||||||
|
|
||||||
|
Problem: Uneinheitliche Schreibweisen / Fehler in den Rednerlisten.
|
||||||
|
|
||||||
|
\pause
|
||||||
|
Lösung: Funktion: \lstinline{repair_speaker(speakers)}
|
||||||
|
\pause
|
||||||
|
\begin{itemize}[<+->]
|
||||||
|
\item Erhält \lstinline{tibble} von Rednern
|
||||||
|
\item Entfernt Redner mit ungültigen, doppelt vergebenen IDs
|
||||||
|
\item Vereinheitlicht Schreibweisen der Fraktionen, Namen und Titel der Redner
|
||||||
|
\end{itemize}
|
||||||
|
|
||||||
|
\end{frame}
|
||||||
|
|
||||||
|
\begin{frame}
|
||||||
|
\frametitle{Reparieren von Fehlern}
|
||||||
|
Problem: Namen in Kommentaren Rednern aus Rednertabelle zuordnen
|
||||||
|
|
||||||
|
\pause
|
||||||
|
Lösung: Funktion \lstinline{repair_comments(comments, speakers)}
|
||||||
|
\begin{itemize}
|
||||||
|
\item Erstellt für jeden Redner einen Regulären Ausdruck aus dem Namen
|
||||||
|
\item Sucht für jeden Kommentar nach dem entsprechenden Eintrag in der
|
||||||
|
Rednertabelle
|
||||||
|
\end{itemize}
|
||||||
|
\end{frame}
|
||||||
|
|
||||||
|
\begin{frame}
|
||||||
|
\frametitle{Reparieren von Fehlern}
|
||||||
|
Beide Reparaturschritte werden in der Funktion \lstinline{repair} zusammengefasst.
|
||||||
|
\end{frame}
|
||||||
|
|
||||||
|
\section{Analyse}
|
||||||
|
|
||||||
|
\begin{frame}
|
||||||
|
\frametitle{Analyse}
|
||||||
|
Stelle Hilfsfunktionen zur Analyse der Daten zur Verfügung:
|
||||||
|
\begin{itemize}[<+->]
|
||||||
|
\item \lstinline{bar_plot_fractions}: Erstellt ein Balkendiagramm aus einer Tabelle
|
||||||
|
mit Fraktionsdaten
|
||||||
|
\item \lstinline{find_word}: Fügt in der Redebeiträgetabelle zu jedem Redebeitrag
|
||||||
|
die Häufigkeit eines Regulären Ausdrucks hinzu.
|
||||||
|
\item \lstinline{word_usage_by_date}: Zählt an welchen Daten (Tagen) ein regulärer Ausdruck
|
||||||
|
wie oft verwendet wird.
|
||||||
|
\item \lstinline{join_speaker}: Fügt einer Tabelle mit Spalte \lstinline{speaker} die
|
||||||
|
enstprechenden Informationen aus der Rednertabelle hinzu.
|
||||||
|
\end{itemize}
|
||||||
|
\end{frame}
|
||||||
|
|
||||||
|
\end{document}
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/analyze.R
|
||||||
|
\name{bar_plot_fractions}
|
||||||
|
\alias{bar_plot_fractions}
|
||||||
|
\title{Bar chart visualizing fraction based data}
|
||||||
|
\usage{
|
||||||
|
bar_plot_fractions(
|
||||||
|
tb,
|
||||||
|
x_variable = NULL,
|
||||||
|
y_variable = NULL,
|
||||||
|
fill = NULL,
|
||||||
|
title = NULL,
|
||||||
|
xlab = "Fraction",
|
||||||
|
ylab = "n",
|
||||||
|
filllab = "Fraction",
|
||||||
|
flipped = TRUE,
|
||||||
|
position = "dodge",
|
||||||
|
reorder = FALSE,
|
||||||
|
rotatelab = FALSE
|
||||||
|
)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{tb}{tibble}
|
||||||
|
|
||||||
|
\item{x_variable}{column in tb, default is fraction}
|
||||||
|
|
||||||
|
\item{y_variable}{column in tb, default is n}
|
||||||
|
|
||||||
|
\item{fill}{column in tb, default is fraction}
|
||||||
|
|
||||||
|
\item{title}{plot title}
|
||||||
|
|
||||||
|
\item{xlab}{label for x axis, default is fraction}
|
||||||
|
|
||||||
|
\item{ylab}{label for y axis, default is n}
|
||||||
|
|
||||||
|
\item{filllab}{default is 'Fraction'}
|
||||||
|
|
||||||
|
\item{flipped}{if TRUE draw bars horizontally, else vertically. Default is TRUE}
|
||||||
|
|
||||||
|
\item{position}{default is 'dodge'}
|
||||||
|
|
||||||
|
\item{reorder}{Either reorder fraction factor by variable value or reorder fraction factor by party seat order in parliament (default).}
|
||||||
|
|
||||||
|
\item{rotatelab}{Default is FALSE. If true turns the labels 90 degrees to the axis.
|
||||||
|
|
||||||
|
plot data from tb in the following way: for each item in x_variable show the corresponding value in y_variable.
|
||||||
|
Then color the plot depending on the fill value.
|
||||||
|
Give the plot a title and a label for x-axis and y-axis,
|
||||||
|
color the legend according to filllab and finally
|
||||||
|
improve positioning details according to position}
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Can be configured to also visualize data not related to fractions.
|
||||||
|
}
|
||||||
+5
-1
@@ -4,10 +4,14 @@
|
|||||||
\alias{fetch_all}
|
\alias{fetch_all}
|
||||||
\title{Download available records}
|
\title{Download available records}
|
||||||
\usage{
|
\usage{
|
||||||
fetch_all(download_dir = "records/", create = FALSE)
|
fetch_all(download_dir = "inst/records/", create = FALSE)
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{download_dir}{character}
|
\item{download_dir}{character}
|
||||||
|
|
||||||
|
\item{create}{bool
|
||||||
|
|
||||||
|
if create is TRUE, the directory given in download_dir is created}
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
This fetches all available records of the 19th legislative period of the german Bundestag.
|
This fetches all available records of the 19th legislative period of the german Bundestag.
|
||||||
|
|||||||
@@ -0,0 +1,18 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/analyze.R
|
||||||
|
\name{find_word}
|
||||||
|
\alias{find_word}
|
||||||
|
\title{Count number of occurences of a given word}
|
||||||
|
\usage{
|
||||||
|
find_word(res, word)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{res}{tibble}
|
||||||
|
|
||||||
|
\item{word}{character
|
||||||
|
|
||||||
|
Add number of occurences of word to talks}
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Count number of occurences of a given word
|
||||||
|
}
|
||||||
@@ -4,15 +4,30 @@
|
|||||||
\name{hateimparlament-package}
|
\name{hateimparlament-package}
|
||||||
\alias{hateimparlament}
|
\alias{hateimparlament}
|
||||||
\alias{hateimparlament-package}
|
\alias{hateimparlament-package}
|
||||||
\title{hateimparlament: Protocolanalysis of German Bundestag}
|
\title{hateimparlament: Recordanalysis Of Bundestag}
|
||||||
\description{
|
\description{
|
||||||
Downloads, parses and analyses protocols of the current German parliament (Bundestag).
|
Downloads, parses and analyses parliamentary records of the 19th legislative
|
||||||
|
period of the German parliament (Bundestag).
|
||||||
}
|
}
|
||||||
\details{
|
\details{
|
||||||
hateimparlament ist ein großartiges Paket!
|
hateimparlament ist ein großartiges Paket!
|
||||||
|
}
|
||||||
|
\seealso{
|
||||||
|
Useful links:
|
||||||
|
\itemize{
|
||||||
|
\item \url{https://git.flavigny.de/christian/hateimparlament}
|
||||||
|
\item Report bugs at \url{https://git.flavigny.de/christian/hateimparlament/issues}
|
||||||
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
\author{
|
\author{
|
||||||
\strong{Maintainer}: First Last \email{first.last@example.com} (\href{https://orcid.org/YOUR-ORCID-ID}{ORCID})
|
\strong{Maintainer}: Christian Merten \email{christian@merten.dev}
|
||||||
|
|
||||||
|
Authors:
|
||||||
|
\itemize{
|
||||||
|
\item Leon Burgard
|
||||||
|
\item Josua Kugler
|
||||||
|
}
|
||||||
|
|
||||||
}
|
}
|
||||||
\keyword{internal}
|
\keyword{internal}
|
||||||
|
|||||||
@@ -0,0 +1,21 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/analyze.R
|
||||||
|
\name{join_speaker}
|
||||||
|
\alias{join_speaker}
|
||||||
|
\title{add information from speaker table to a tibble containing speaker id}
|
||||||
|
\usage{
|
||||||
|
join_speaker(tb, res, fraction_only = F)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{tb}{tibble}
|
||||||
|
|
||||||
|
\item{res}{list of tibbles}
|
||||||
|
|
||||||
|
\item{fraction_only}{if TRUE, only select fraction from the resulting joined tibble
|
||||||
|
|
||||||
|
left join speaker information from res$speaker into tb.
|
||||||
|
if fraction_only, drop all columns but fraction}
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
add information from speaker table to a tibble containing speaker id
|
||||||
|
}
|
||||||
@@ -0,0 +1,24 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/repair.R
|
||||||
|
\name{lookup_speaker}
|
||||||
|
\alias{lookup_speaker}
|
||||||
|
\title{Lookup name in speakers table}
|
||||||
|
\usage{
|
||||||
|
lookup_speaker(tb, speaker, name_variable)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{tb}{tibble}
|
||||||
|
|
||||||
|
\item{speaker}{tibble}
|
||||||
|
|
||||||
|
\item{name_variable}{name
|
||||||
|
|
||||||
|
Tries to match the name_variable column with speaker names
|
||||||
|
|
||||||
|
returns a lookup table}
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Tries to find the correct speaker id given a name.
|
||||||
|
This is sufficient since every prename lastname combination in the bundestag is
|
||||||
|
unique (luckily :D)
|
||||||
|
}
|
||||||
@@ -0,0 +1,16 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/analyze.R
|
||||||
|
\docType{data}
|
||||||
|
\name{party_colors}
|
||||||
|
\alias{party_colors}
|
||||||
|
\title{lookup table for official party colors}
|
||||||
|
\format{
|
||||||
|
An object of class \code{character} of length 8.
|
||||||
|
}
|
||||||
|
\usage{
|
||||||
|
party_colors
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
lookup table for official party colors
|
||||||
|
}
|
||||||
|
\keyword{datasets}
|
||||||
+4
-2
@@ -4,10 +4,12 @@
|
|||||||
\alias{read_all}
|
\alias{read_all}
|
||||||
\title{Parse xml records}
|
\title{Parse xml records}
|
||||||
\usage{
|
\usage{
|
||||||
read_all(path = "records/")
|
read_all(path = "inst/records/", pattern = "-data\\\\.xml")
|
||||||
}
|
}
|
||||||
\arguments{
|
\arguments{
|
||||||
\item{path}{character}
|
\item{path}{path to records directory}
|
||||||
|
|
||||||
|
\item{pattern}{search pattern to find records in directory}
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
Creates a list of tibbles containing relevant information from all records
|
Creates a list of tibbles containing relevant information from all records
|
||||||
|
|||||||
@@ -0,0 +1,16 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/parse.R
|
||||||
|
\name{read_from_csv}
|
||||||
|
\alias{read_from_csv}
|
||||||
|
\title{create a tibble from the csv file}
|
||||||
|
\usage{
|
||||||
|
read_from_csv(path = "inst/csv/")
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{path}{directory to read files from
|
||||||
|
|
||||||
|
reading the tables from a csv is way faster than reading and repairing the data every single time}
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
create a tibble from the csv file
|
||||||
|
}
|
||||||
+11
-1
@@ -4,7 +4,17 @@
|
|||||||
\alias{repair}
|
\alias{repair}
|
||||||
\title{Repair parsed tables}
|
\title{Repair parsed tables}
|
||||||
\usage{
|
\usage{
|
||||||
repair(parse_output)
|
repair(parse_output, lookup_speaker = FALSE)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{parse_output}{tibble}
|
||||||
|
|
||||||
|
\item{lookup_speaker}{bool
|
||||||
|
|
||||||
|
If lookup_speaker is TRUE, members of the parliament mentioned in comments are looked up in speaker table.
|
||||||
|
|
||||||
|
Possible test: check identical(repair(res), repair(repair(res))) == TRUE
|
||||||
|
Since repaired tables should be a fixpoint of repair.}
|
||||||
}
|
}
|
||||||
\description{
|
\description{
|
||||||
Repair parsed tables
|
Repair parsed tables
|
||||||
|
|||||||
@@ -0,0 +1,18 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/analyze.R
|
||||||
|
\name{word_usage_by_date}
|
||||||
|
\alias{word_usage_by_date}
|
||||||
|
\title{Word usage summarised by date}
|
||||||
|
\usage{
|
||||||
|
word_usage_by_date(res, patterns, tidy = F)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{res}{List of Tibbles to be analysed.}
|
||||||
|
|
||||||
|
\item{patterns}{Words to look up.}
|
||||||
|
|
||||||
|
\item{tidy}{default is FALSE.}
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Counts how many talks do match a given pattern and summarises by date.
|
||||||
|
}
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
% Generated by roxygen2: do not edit by hand
|
||||||
|
% Please edit documentation in R/parse.R
|
||||||
|
\name{write_to_csv}
|
||||||
|
\alias{write_to_csv}
|
||||||
|
\title{Write the parsed and repaired results into separate csv files}
|
||||||
|
\usage{
|
||||||
|
write_to_csv(tables, path = "inst/csv/", create = F)
|
||||||
|
}
|
||||||
|
\arguments{
|
||||||
|
\item{tables}{list of tables to convert into a csv files.}
|
||||||
|
|
||||||
|
\item{path}{where to put the csv files.}
|
||||||
|
|
||||||
|
\item{create}{set TRUE if the path does not exist yet and you want to create it}
|
||||||
|
}
|
||||||
|
\description{
|
||||||
|
Write the parsed and repaired results into separate csv files
|
||||||
|
}
|
||||||
@@ -0,0 +1,83 @@
|
|||||||
|
---
|
||||||
|
title: "Analysis of covered topics"
|
||||||
|
output: rmarkdown::html_vignette
|
||||||
|
vignette: >
|
||||||
|
%\VignetteIndexEntry{Analysis of covered topics}
|
||||||
|
%\VignetteEngine{knitr::rmarkdown}
|
||||||
|
%\VignetteEncoding{UTF-8}
|
||||||
|
---
|
||||||
|
|
||||||
|
```{r, include = FALSE}
|
||||||
|
knitr::opts_chunk$set(
|
||||||
|
collapse = TRUE,
|
||||||
|
comment = "#>"
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
```{r setup}
|
||||||
|
library(hateimparlament)
|
||||||
|
library(dplyr)
|
||||||
|
library(ggplot2)
|
||||||
|
library(stringr)
|
||||||
|
library(tidyr)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Preparation of data
|
||||||
|
|
||||||
|
First, you need to download all records of the current legislative period.
|
||||||
|
```r
|
||||||
|
fetch_all("../inst/records/") # path to directory where records should be stored
|
||||||
|
```
|
||||||
|
Second, those `.xml` files, need to be parsed into `R` `tibbles`. This is accomplished by:
|
||||||
|
```r
|
||||||
|
read_all("../inst/records/") %>% repair() -> res
|
||||||
|
```
|
||||||
|
We also used `repair` to fix a bunch of formatting issues in the records.
|
||||||
|
|
||||||
|
For development purposes, we load the tables from csv files.
|
||||||
|
```{r}
|
||||||
|
res <- read_from_csv('../inst/csv/')
|
||||||
|
```
|
||||||
|
|
||||||
|
## Analysis
|
||||||
|
|
||||||
|
Now we can start analysing our parsed dataset:
|
||||||
|
|
||||||
|
### Counting the occurences of a given word:
|
||||||
|
|
||||||
|
```{r, fig.width=7, fig.height=7}
|
||||||
|
find_word(res, "Kohleausstieg") %>%
|
||||||
|
filter(occurences > 0) %>%
|
||||||
|
join_speaker(res) %>%
|
||||||
|
select(content, fraction) %>%
|
||||||
|
filter(!is.na(fraction)) %>%
|
||||||
|
group_by(fraction) %>%
|
||||||
|
summarize(n = n()) %>%
|
||||||
|
arrange(desc(n)) %>%
|
||||||
|
bar_plot_fractions(title = "Parties using the word 'Kohleausstieg' the most (absolutely)",
|
||||||
|
ylab = "Number of uses of 'Kohleausstieg'",
|
||||||
|
flipped = F,
|
||||||
|
rotatelab = T)
|
||||||
|
```
|
||||||
|
|
||||||
|
### When are which topics discussed the most?
|
||||||
|
|
||||||
|
First we define some search patterns, according to some common political topics.
|
||||||
|
```{r}
|
||||||
|
pandemic_pattern <- "(?i)virus|corona|covid|lockdown"
|
||||||
|
climate_pattern <- "(?i)klimawandel|erderwärmung|co2|treibhaus|methan|kyoto-protokoll|klimaabkommen"
|
||||||
|
pension_pattern <- "(?i)rente|pension|altersarmut"
|
||||||
|
```
|
||||||
|
Then we use the analysis helper `word_usage_by_date` to generate a tibble counting the
|
||||||
|
occurences of our search patterns per date. We can then plot the results:
|
||||||
|
```{r, fig.width=7, fig.height=6}
|
||||||
|
word_usage_by_date(res, c(pandemic = pandemic_pattern,
|
||||||
|
climate = climate_pattern,
|
||||||
|
pension = pension_pattern)) %>%
|
||||||
|
ggplot(aes(x = date, y = count, color = pattern)) +
|
||||||
|
xlab("date of session") +
|
||||||
|
ylab("occurence of word per session") +
|
||||||
|
labs(color = "Topic") +
|
||||||
|
geom_point()
|
||||||
|
```
|
||||||
|
|
||||||
@@ -1,70 +0,0 @@
|
|||||||
---
|
|
||||||
title: "funwithdata"
|
|
||||||
output: rmarkdown::html_vignette
|
|
||||||
vignette: >
|
|
||||||
%\VignetteIndexEntry{funwithdata}
|
|
||||||
%\VignetteEngine{knitr::rmarkdown}
|
|
||||||
%\VignetteEncoding{UTF-8}
|
|
||||||
---
|
|
||||||
|
|
||||||
```{r, include = FALSE}
|
|
||||||
knitr::opts_chunk$set(
|
|
||||||
collapse = TRUE,
|
|
||||||
comment = "#>"
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
```{r setup}
|
|
||||||
library(hateimparlament)
|
|
||||||
library(dplyr)
|
|
||||||
library(ggplot2)
|
|
||||||
```
|
|
||||||
|
|
||||||
## Preparation of data
|
|
||||||
|
|
||||||
First, you need to download all records of the current legislative period.
|
|
||||||
```r
|
|
||||||
fetch_all("../records/") # path to directory where records should be stored
|
|
||||||
```
|
|
||||||
Second, those `.xml` files, need to be parsed into `R` `tibbles`. This is accomplished by:
|
|
||||||
```r
|
|
||||||
read_all("../records/") %>% repair() -> res
|
|
||||||
```
|
|
||||||
We also used `repair` to fix a bunch of formatting issues in the records and unpacked
|
|
||||||
the result into more descriptive variables.
|
|
||||||
|
|
||||||
For development purposes, we load the tables from csv files.
|
|
||||||
```{r}
|
|
||||||
res <- read_from_csv('../csv/')
|
|
||||||
```
|
|
||||||
and unpack our tibbles
|
|
||||||
```{r}
|
|
||||||
comments <- res$comments
|
|
||||||
reden <- res$reden
|
|
||||||
redner <- res$redner
|
|
||||||
talks <- res$talks
|
|
||||||
```
|
|
||||||
|
|
||||||
## Analysis
|
|
||||||
|
|
||||||
Now we can start analysing our parsed dataset, e.g. find out which party gives the most talks:
|
|
||||||
```{r, fig.width=10}
|
|
||||||
join_redner(reden, res) %>%
|
|
||||||
group_by(fraktion) %>%
|
|
||||||
summarize(n = n()) %>%
|
|
||||||
arrange(n) %>%
|
|
||||||
bar_plot_fraktionen()
|
|
||||||
```
|
|
||||||
|
|
||||||
### Count a word occurence
|
|
||||||
|
|
||||||
```{r, fig.width=10}
|
|
||||||
find_word(res, "hitler") %>%
|
|
||||||
filter(occurences > 0) %>%
|
|
||||||
join_redner(res) %>%
|
|
||||||
select(content, fraktion) %>%
|
|
||||||
group_by(fraktion) %>%
|
|
||||||
summarize(n = n()) %>%
|
|
||||||
arrange(desc(n)) %>%
|
|
||||||
bar_plot_fraktionen()
|
|
||||||
```
|
|
||||||
@@ -0,0 +1,195 @@
|
|||||||
|
---
|
||||||
|
title: "Differences in gender"
|
||||||
|
output: rmarkdown::html_vignette
|
||||||
|
vignette: >
|
||||||
|
%\VignetteIndexEntry{Differences in gender}
|
||||||
|
%\VignetteEngine{knitr::rmarkdown}
|
||||||
|
%\VignetteEncoding{UTF-8}
|
||||||
|
---
|
||||||
|
|
||||||
|
```{r, include = FALSE}
|
||||||
|
knitr::opts_chunk$set(
|
||||||
|
collapse = TRUE,
|
||||||
|
comment = "#>"
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
```{r setup}
|
||||||
|
library(hateimparlament)
|
||||||
|
library(dplyr)
|
||||||
|
library(ggplot2)
|
||||||
|
library(stringr)
|
||||||
|
library(tidyr)
|
||||||
|
library(xml2)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Preparation of data
|
||||||
|
|
||||||
|
First, you need to download all records of the current legislative period.
|
||||||
|
```r
|
||||||
|
fetch_all("../records/") # path to directory where records should be stored
|
||||||
|
```
|
||||||
|
Second, those `.xml` files, need to be parsed into `R` `tibbles`. This is accomplished by:
|
||||||
|
```r
|
||||||
|
read_all("../records/") %>% repair() -> res
|
||||||
|
```
|
||||||
|
We also used `repair` to fix a bunch of formatting issues in the records.
|
||||||
|
|
||||||
|
For development purposes, we load the tables from csv files.
|
||||||
|
```{r}
|
||||||
|
res <- read_from_csv('../inst/csv/')
|
||||||
|
```
|
||||||
|
and unpack our tibbles
|
||||||
|
```{r}
|
||||||
|
comments <- res$comments
|
||||||
|
speeches <- res$speeches
|
||||||
|
speaker <- res$speaker
|
||||||
|
talks <- res$talks
|
||||||
|
```
|
||||||
|
|
||||||
|
Bevor we can do our analysis, we have to assign a gender to our politicans. We do this
|
||||||
|
by reading the gender from the master data of all members of parliament, which is
|
||||||
|
fetched from bundestag.de.
|
||||||
|
|
||||||
|
```{r}
|
||||||
|
xml_get <- function(node, name) {
|
||||||
|
res <- xml_text(xml_find_all(node, name))
|
||||||
|
if (length(res) == 0) NA_character_
|
||||||
|
else res
|
||||||
|
}
|
||||||
|
|
||||||
|
x <- read_xml("../inst/masterdata.xml")
|
||||||
|
mdbs <- xml_find_all(x, "MDB")
|
||||||
|
|
||||||
|
ids <- c()
|
||||||
|
genders <- c()
|
||||||
|
for (mdb in mdbs) {
|
||||||
|
xml_get(mdb, "ID") -> mdb_id
|
||||||
|
xml_find_first(mdb, "BIOGRAFISCHE_ANGABEN") %>%
|
||||||
|
xml_get("GESCHLECHT") ->
|
||||||
|
mdb_gender
|
||||||
|
ids <- c(ids, mdb_id)
|
||||||
|
genders <- c(genders, if (mdb_gender == "männlich") "male" else "female")
|
||||||
|
}
|
||||||
|
|
||||||
|
gender <- tibble(id = ids, gender = genders)
|
||||||
|
speaker_with_gender <- left_join(res$speaker, gender)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Analyse
|
||||||
|
|
||||||
|
First, let's look at the relative distribution of the sexes throughout the whole Bundestag.
|
||||||
|
|
||||||
|
```{r}
|
||||||
|
speaker_with_gender %>%
|
||||||
|
select(gender) %>%
|
||||||
|
group_by(gender) %>%
|
||||||
|
summarise("count" = n()) %>%
|
||||||
|
filter(gender %in% c("male", "female")) %>%
|
||||||
|
mutate(portion = 100*count/sum(count)) ->
|
||||||
|
plot1
|
||||||
|
|
||||||
|
bp <- ggplot(plot1, aes(x = "", y = portion, fill = gender))+
|
||||||
|
geom_bar(width = 1, stat = "identity")
|
||||||
|
pie <- bp + coord_polar("y", start=0)
|
||||||
|
pie +
|
||||||
|
scale_fill_manual(values=c("pink", "blue")) +
|
||||||
|
ggtitle("Relative distribution of sexes") +
|
||||||
|
xlab("") +
|
||||||
|
ylab("")
|
||||||
|
```
|
||||||
|
|
||||||
|
Next, we look at the individual distributions between men and women in the different fractions.
|
||||||
|
|
||||||
|
```{r, fig.width=7}
|
||||||
|
speaker_with_gender %>%
|
||||||
|
group_by(fraction) %>%
|
||||||
|
summarize(n = n()) ->
|
||||||
|
fraction_size
|
||||||
|
|
||||||
|
speaker_with_gender %>%
|
||||||
|
filter(gender=="female") %>%
|
||||||
|
group_by(fraction) %>%
|
||||||
|
summarize(n_female = n()) %>%
|
||||||
|
left_join(fraction_size) %>%
|
||||||
|
mutate(q = n_female/n) -> women_per_fraction
|
||||||
|
bar_plot_fractions(women_per_fraction, x_variable=fraction, y_variable=q, title="Frauenanteil nach Partei")
|
||||||
|
```
|
||||||
|
|
||||||
|
Prepared with this knowledge, we can now analyse the relative amount of speeches by gender and fraction.
|
||||||
|
|
||||||
|
```{r, fig.width=7}
|
||||||
|
speaker_with_gender %>% transmute(speaker_id = id, gender, fraction) -> simple_speaker_with_gender
|
||||||
|
speeches %>%
|
||||||
|
transmute(id, speaker_id = speaker) %>%
|
||||||
|
inner_join(simple_speaker_with_gender) %>%
|
||||||
|
group_by(fraction) %>%
|
||||||
|
summarize(speeches=n()) ->
|
||||||
|
fraction_speeches_size
|
||||||
|
|
||||||
|
speeches %>%
|
||||||
|
transmute(id, speaker_id = speaker) %>%
|
||||||
|
inner_join(simple_speaker_with_gender) %>%
|
||||||
|
filter(gender=='female') %>%
|
||||||
|
group_by(fraction) %>%
|
||||||
|
summarize(female_speeches=n()) %>%
|
||||||
|
left_join(fraction_speeches_size) %>%
|
||||||
|
left_join(women_per_fraction) %>%
|
||||||
|
mutate(q_speeches = female_speeches/speeches) -> speech_distribution
|
||||||
|
#bar_plot_fractions(speech_distribution, x_variable=fraction, y_variable=q_speeches, title="Redeanteil Frauen nach Partei")
|
||||||
|
|
||||||
|
|
||||||
|
party_order <- factor(c("Fraktionslos", "AfD&Fraktionslos",
|
||||||
|
"DIE LINKE", "BÜNDNIS 90/DIE GRÜNEN", "SPD", "CDU/CSU",
|
||||||
|
"FDP", "AfD", NA_character_))
|
||||||
|
|
||||||
|
speech_distribution %>%
|
||||||
|
mutate("Frauenanteil" = q, "Redenanteil Frauen" = q_speeches) %>%
|
||||||
|
pivot_longer(c(Frauenanteil, "Redenanteil Frauen"), "type") %>%
|
||||||
|
ggplot(aes(x=factor(fraction, levels = party_order), y=value, fill=factor(type, levels = factor(c("Frauenanteil", "Redenanteil Frauen"))))) + scale_fill_manual(values= c("Frauenanteil"="gray", "Redenanteil Frauen"="red")) + coord_flip() + geom_bar(stat="identity", position="dodge") + labs(fill="Kategorie")
|
||||||
|
|
||||||
|
```
|
||||||
|
|
||||||
|
For comparison, let's analyze the total differences in the amount of speeches given.
|
||||||
|
```{r}
|
||||||
|
|
||||||
|
speeches %>%
|
||||||
|
group_by(speaker) %>%
|
||||||
|
summarize(n = n()) %>%
|
||||||
|
ungroup() %>%
|
||||||
|
arrange(-n) %>%
|
||||||
|
join_speaker(res) %>%
|
||||||
|
left_join(gender, by=c("speaker"="id")) %>%
|
||||||
|
group_by(gender) %>%
|
||||||
|
summarise(absolute=sum(n)) %>%
|
||||||
|
filter(gender %in% c("female", "male")) %>%
|
||||||
|
mutate(absolute2=absolute/sum(absolute)) %>%
|
||||||
|
mutate(portion=c(0.32, 0.68)) %>%
|
||||||
|
mutate(relative=absolute*(1-portion)) %>%
|
||||||
|
mutate(relative2=relative/sum(relative)) ->
|
||||||
|
plot3
|
||||||
|
```
|
||||||
|
|
||||||
|
At first lets take a look at the absolute difference in the amount of speeches by the two sexes.
|
||||||
|
```{r,fig.width=7}
|
||||||
|
barplot(plot3$absolute2,
|
||||||
|
ylab = "amount of speeches",
|
||||||
|
main = "Absolute comparison of speech shares",
|
||||||
|
las = 1,
|
||||||
|
names.arg = c("women", "men"),
|
||||||
|
col = c("pink", "darkblue"),
|
||||||
|
font.main = 4,
|
||||||
|
cex.axis = 0.7)
|
||||||
|
```
|
||||||
|
|
||||||
|
Since there are more men represented in the German Bundestag, we now consider the relative proportions of speeches, depending on the ratio of men and women.
|
||||||
|
```{r, fig.width=7}
|
||||||
|
barplot(plot3$relative2,
|
||||||
|
ylab = "amount of speeches",
|
||||||
|
main = "Relative comparison of speech shares",
|
||||||
|
las = 1,
|
||||||
|
names.arg = c("women", "men"),
|
||||||
|
col = c("pink", "darkblue"),
|
||||||
|
font.main = 4,
|
||||||
|
cex.axis = 0.7)
|
||||||
|
```
|
||||||
@@ -0,0 +1,82 @@
|
|||||||
|
---
|
||||||
|
title: "General questions"
|
||||||
|
output: rmarkdown::html_vignette
|
||||||
|
vignette: >
|
||||||
|
%\VignetteIndexEntry{General questions}
|
||||||
|
%\VignetteEngine{knitr::rmarkdown}
|
||||||
|
%\VignetteEncoding{UTF-8}
|
||||||
|
---
|
||||||
|
|
||||||
|
```{r, include = FALSE}
|
||||||
|
knitr::opts_chunk$set(
|
||||||
|
collapse = TRUE,
|
||||||
|
comment = "#>"
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
```{r setup}
|
||||||
|
library(hateimparlament)
|
||||||
|
library(dplyr)
|
||||||
|
library(ggplot2)
|
||||||
|
library(stringr)
|
||||||
|
library(tidyr)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Preparation of data
|
||||||
|
|
||||||
|
First, you need to download all records of the current legislative period.
|
||||||
|
```r
|
||||||
|
fetch_all("../inst/records/") # path to directory where records should be stored
|
||||||
|
```
|
||||||
|
Second, those `.xml` files, need to be parsed into `R` `tibbles`. This is accomplished by:
|
||||||
|
```r
|
||||||
|
read_all("../inst/records/") %>% repair() -> res
|
||||||
|
```
|
||||||
|
We also used `repair` to fix a bunch of formatting issues in the records.
|
||||||
|
|
||||||
|
For development purposes, we load the tables from csv files.
|
||||||
|
```{r}
|
||||||
|
res <- read_from_csv('../inst/csv/')
|
||||||
|
```
|
||||||
|
|
||||||
|
## Analysis
|
||||||
|
|
||||||
|
Now we can start analysing our parsed dataset:
|
||||||
|
|
||||||
|
### Which party gives the most talks?
|
||||||
|
|
||||||
|
```{r, fig.width=7}
|
||||||
|
join_speaker(res$speeches, res) %>%
|
||||||
|
group_by(fraction) %>%
|
||||||
|
summarize(n = n()) %>%
|
||||||
|
arrange(n) %>%
|
||||||
|
bar_plot_fractions(title="Number of speeches given by fraction",
|
||||||
|
ylab="Number of speeches")
|
||||||
|
```
|
||||||
|
|
||||||
|
Note that `NA` signifies speeches given by speakers who are not members of parliament.
|
||||||
|
|
||||||
|
### Who gives the most speeches?
|
||||||
|
|
||||||
|
```{r}
|
||||||
|
res$speeches %>%
|
||||||
|
group_by(speaker) %>%
|
||||||
|
summarize(n = n()) %>%
|
||||||
|
arrange(-n) %>%
|
||||||
|
left_join(res$speaker, by=c("speaker" = "id")) %>%
|
||||||
|
head(10)
|
||||||
|
```
|
||||||
|
|
||||||
|
### Who talks the longest?
|
||||||
|
|
||||||
|
Calculate the average character length of talks given by speakers:
|
||||||
|
|
||||||
|
```{r}
|
||||||
|
res$talks %>%
|
||||||
|
mutate(content_len = str_length(content)) %>%
|
||||||
|
group_by(speaker) %>%
|
||||||
|
summarize(avg_content_len = mean(content_len)) %>%
|
||||||
|
arrange(-avg_content_len) %>%
|
||||||
|
left_join(res$speaker, by=c("speaker" = "id")) %>%
|
||||||
|
head(10)
|
||||||
|
```
|
||||||
@@ -0,0 +1,187 @@
|
|||||||
|
---
|
||||||
|
title: "Analysis of vocabulary"
|
||||||
|
output: rmarkdown::html_vignette
|
||||||
|
vignette: >
|
||||||
|
%\VignetteIndexEntry{Analysis of vocabulary}
|
||||||
|
%\VignetteEngine{knitr::rmarkdown}
|
||||||
|
%\VignetteEncoding{UTF-8}
|
||||||
|
---
|
||||||
|
|
||||||
|
```{r, include = FALSE}
|
||||||
|
knitr::opts_chunk$set(
|
||||||
|
collapse = TRUE,
|
||||||
|
comment = "#>"
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
```{r setup}
|
||||||
|
library(hateimparlament)
|
||||||
|
library(dplyr)
|
||||||
|
library(stringr)
|
||||||
|
library(ggplot2)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Preparation of data
|
||||||
|
|
||||||
|
First, you need to download all records of the current legislative period.
|
||||||
|
```r
|
||||||
|
fetch_all("../inst/records/") # path to directory where records should be stored
|
||||||
|
```
|
||||||
|
Second, those `.xml` files, need to be parsed into `R` `tibbles`. This is accomplished by:
|
||||||
|
```r
|
||||||
|
read_all("../inst/records/") %>% repair() -> res
|
||||||
|
|
||||||
|
speeches <- res$speeches
|
||||||
|
speaker <- res$speaker
|
||||||
|
talks <- res$talks
|
||||||
|
```
|
||||||
|
We also used `repair` to fix a bunch of formatting issues in the records and unpacked
|
||||||
|
the result into more descriptive variables.
|
||||||
|
|
||||||
|
For development purposes, we load the tables from csv files.
|
||||||
|
```{r}
|
||||||
|
tables <- read_from_csv('../inst/csv/')
|
||||||
|
|
||||||
|
comments <- tables$comments
|
||||||
|
speeches <- tables$speeches
|
||||||
|
speaker <- tables$speaker
|
||||||
|
talks <- tables$talks
|
||||||
|
```
|
||||||
|
|
||||||
|
Further, we need to load a list of words that were used by Hitler but not by standard German texts.
|
||||||
|
```{r}
|
||||||
|
fil <- file('../inst/hitler_texts/hitler_words')
|
||||||
|
Worte <- readLines(fil)
|
||||||
|
hitlerwords <- tibble(Worte)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Analysis
|
||||||
|
|
||||||
|
Now we extract the words that were used with higher frequency by one party and compare them with `hitlerwords`.
|
||||||
|
```{r}
|
||||||
|
talks %>%
|
||||||
|
left_join(speaker, by=c(speaker='id')) %>%
|
||||||
|
group_by(fraction) %>%
|
||||||
|
summarize(full_text=str_c(content, collapse="\n")) -> talks_by_fraction
|
||||||
|
```
|
||||||
|
For each party, we want to get a tibble of words with frequency.
|
||||||
|
```{r}
|
||||||
|
#AfD
|
||||||
|
Worte <- str_extract_all(talks_by_fraction$full_text[[1]], "\\b[a-zA-ZäöüÄÖÜß]+\\b")[[1]]
|
||||||
|
afdtotal = length(Worte)
|
||||||
|
tibble(Worte) %>% group_by(Worte) %>% count() %>% mutate(freq =n/afdtotal) -> afd_words
|
||||||
|
|
||||||
|
#AfD&Fraktionslos
|
||||||
|
Worte <- str_extract_all(talks_by_fraction$full_text[[2]], "\\b[a-zA-ZäöüÄÖÜß]+\\b")[[1]]
|
||||||
|
afdundfraktionslostotal = length(Worte)
|
||||||
|
tibble(Worte) %>% group_by(Worte) %>% count() %>% mutate(freq =n/afdundfraktionslostotal) -> afdundfraktionslos_words
|
||||||
|
|
||||||
|
#BÜNDNIS 90 / DIE GRÜNEN
|
||||||
|
Worte <- str_extract_all(talks_by_fraction$full_text[[3]], "\\b[a-zA-ZäöüÄÖÜß]+\\b")[[1]]
|
||||||
|
grünetotal = length(Worte)
|
||||||
|
tibble(Worte) %>% group_by(Worte) %>% count() %>% mutate(freq =n/grünetotal) -> grüne_words
|
||||||
|
|
||||||
|
#CDU/CSU
|
||||||
|
Worte <- str_extract_all(talks_by_fraction$full_text[[4]], "\\b[a-zA-ZäöüÄÖÜß]+\\b")[[1]]
|
||||||
|
cdutotal = length(Worte)
|
||||||
|
tibble(Worte) %>% group_by(Worte) %>% count() %>% mutate(freq =n/cdutotal) -> cdu_words
|
||||||
|
|
||||||
|
#DIE LINKE
|
||||||
|
Worte <- str_extract_all(talks_by_fraction$full_text[[5]], "\\b[a-zA-ZäöüÄÖÜß]+\\b")[[1]]
|
||||||
|
linketotal = length(Worte)
|
||||||
|
tibble(Worte) %>% group_by(Worte) %>% count() %>% mutate(freq =n/linketotal) -> linke_words
|
||||||
|
|
||||||
|
#FDP
|
||||||
|
Worte <- str_extract_all(talks_by_fraction$full_text[[6]], "\\b[a-zA-ZäöüÄÖÜß]+\\b")[[1]]
|
||||||
|
fdptotal = length(Worte)
|
||||||
|
tibble(Worte) %>% group_by(Worte) %>% count() %>% mutate(freq =n/fdptotal) -> fdp_words
|
||||||
|
|
||||||
|
#Fraktionslos
|
||||||
|
Worte <- str_extract_all(talks_by_fraction$full_text[[7]], "\\b[a-zA-ZäöüÄÖÜß]+\\b")[[1]]
|
||||||
|
fraktionslostotal = length(Worte)
|
||||||
|
tibble(Worte) %>% group_by(Worte) %>% count() %>% mutate(freq =n/fraktionslostotal) -> fraktionslos_words
|
||||||
|
|
||||||
|
#SPD
|
||||||
|
Worte <- str_extract_all(talks_by_fraction$full_text[[8]], "\\b[a-zA-ZäöüÄÖÜß]+\\b")[[1]]
|
||||||
|
spdtotal = length(Worte)
|
||||||
|
tibble(Worte) %>% group_by(Worte) %>% count() %>% mutate(freq =n/spdtotal) -> spd_words
|
||||||
|
|
||||||
|
#NA
|
||||||
|
Worte <- str_extract_all(talks_by_fraction$full_text[[9]], "\\b[a-zA-ZäöüÄÖÜß]+\\b")[[1]]
|
||||||
|
natotal = length(Worte)
|
||||||
|
tibble(Worte) %>% group_by(Worte) %>% count() %>% mutate(freq =n/natotal) -> na_words
|
||||||
|
|
||||||
|
#alle
|
||||||
|
all_words <- bind_rows(afd_words, afdundfraktionslos_words, grüne_words, cdu_words, linke_words, fdp_words, fraktionslos_words, spd_words, na_words)
|
||||||
|
total <- sum(all_words$n)
|
||||||
|
all_words %>% group_by(Worte) %>% summarize(n = sum(n), part= sum(n)/total) -> all_words
|
||||||
|
```
|
||||||
|
|
||||||
|
Now we want to extract the words that are more frequently used by a specific fraction.
|
||||||
|
```{r}
|
||||||
|
afd_words %>%
|
||||||
|
transmute(freq, fraction_n = n) %>%
|
||||||
|
left_join(all_words) %>%
|
||||||
|
transmute(
|
||||||
|
fraction_freq = freq,
|
||||||
|
total_freq = part,
|
||||||
|
fraction_n,
|
||||||
|
total_n = n,
|
||||||
|
rel_quotient = fraction_freq/total_freq,
|
||||||
|
abs_quotient = fraction_n/total_n) %>%
|
||||||
|
arrange(-abs_quotient, -fraction_n) %>%
|
||||||
|
filter(rel_quotient > 1) ->
|
||||||
|
afd_high_frequent
|
||||||
|
|
||||||
|
select(afd_high_frequent, fraction_n, total_n, abs_quotient, rel_quotient) %>%
|
||||||
|
filter(total_n > 80)
|
||||||
|
|
||||||
|
afdundfraktionslos_words %>% transmute(freq, fraction_n = n) %>% left_join(all_words) %>% transmute(fraction_freq = freq, total_freq = part, fraction_n, total_n = n, rel_quotient = fraction_freq/total_freq, abs_quotient = fraction_n/total_n) %>% arrange(-abs_quotient, -fraction_n) %>% filter(rel_quotient > 1) -> afdundfraktionslos_high_frequent
|
||||||
|
select(afdundfraktionslos_high_frequent, fraction_n, total_n, abs_quotient, rel_quotient) %>% filter(total_n > 80)
|
||||||
|
|
||||||
|
grüne_words %>% transmute(freq, fraction_n = n) %>% left_join(all_words) %>% transmute(fraction_freq = freq, total_freq = part, fraction_n, total_n = n, rel_quotient = fraction_freq/total_freq, abs_quotient = fraction_n/total_n) %>% arrange(-abs_quotient, -fraction_n) %>% filter(rel_quotient > 1) -> grüne_high_frequent
|
||||||
|
select(grüne_high_frequent, fraction_n, total_n, abs_quotient, rel_quotient) %>% filter(total_n > 80)
|
||||||
|
|
||||||
|
cdu_words %>% transmute(freq, fraction_n = n) %>% left_join(all_words) %>% transmute(fraction_freq = freq, total_freq = part, fraction_n, total_n = n, rel_quotient = fraction_freq/total_freq, abs_quotient = fraction_n/total_n) %>% arrange(-abs_quotient, -fraction_n) %>% filter(rel_quotient > 1) -> cdu_high_frequent
|
||||||
|
select(cdu_high_frequent, fraction_n, total_n, abs_quotient, rel_quotient) %>% filter(total_n > 80)
|
||||||
|
|
||||||
|
linke_words %>% transmute(freq, fraction_n = n) %>% left_join(all_words) %>% transmute(fraction_freq = freq, total_freq = part, fraction_n, total_n = n, rel_quotient = fraction_freq/total_freq, abs_quotient = fraction_n/total_n) %>% arrange(-abs_quotient, -fraction_n) %>% filter(rel_quotient > 1) -> linke_high_frequent
|
||||||
|
select(linke_high_frequent, fraction_n, total_n, abs_quotient, rel_quotient) %>% filter(total_n > 80)
|
||||||
|
|
||||||
|
fdp_words %>% transmute(freq, fraction_n = n) %>% left_join(all_words) %>% transmute(fraction_freq = freq, total_freq = part, fraction_n, total_n = n, rel_quotient = fraction_freq/total_freq, abs_quotient = fraction_n/total_n) %>% arrange(-abs_quotient, -fraction_n) %>% filter(rel_quotient > 1) -> fdp_high_frequent
|
||||||
|
select(fdp_high_frequent, fraction_n, total_n, abs_quotient, rel_quotient) %>% filter(total_n > 80)
|
||||||
|
|
||||||
|
fraktionslos_words %>% transmute(freq, fraction_n = n) %>% left_join(all_words) %>% transmute(fraction_freq = freq, total_freq = part, fraction_n, total_n = n, rel_quotient = fraction_freq/total_freq, abs_quotient = fraction_n/total_n) %>% arrange(-abs_quotient, -fraction_n) %>% filter(rel_quotient > 1) -> fraktionslos_high_frequent
|
||||||
|
select(fraktionslos_high_frequent, fraction_n, total_n, abs_quotient, rel_quotient) %>% filter(total_n > 80)
|
||||||
|
|
||||||
|
spd_words %>% transmute(freq, fraction_n = n) %>% left_join(all_words) %>% transmute(fraction_freq = freq, total_freq = part, fraction_n, total_n = n, rel_quotient = fraction_freq/total_freq, abs_quotient = fraction_n/total_n) %>% arrange(-abs_quotient, -fraction_n) %>% filter(rel_quotient > 1) -> spd_high_frequent
|
||||||
|
select(spd_high_frequent, fraction_n, total_n, abs_quotient, rel_quotient) %>% filter(total_n > 80)
|
||||||
|
|
||||||
|
na_words %>% transmute(freq, fraction_n = n) %>% left_join(all_words) %>% transmute(fraction_freq = freq, total_freq = part, fraction_n, total_n = n, rel_quotient = fraction_freq/total_freq, abs_quotient = fraction_n/total_n) %>% arrange(-abs_quotient, -fraction_n) %>% filter(rel_quotient > 1) -> na_high_frequent
|
||||||
|
select(na_high_frequent, fraction_n, total_n, abs_quotient, rel_quotient) %>% filter(total_n > 80)
|
||||||
|
```
|
||||||
|
|
||||||
|
We compare these words with `hitlerwords`.
|
||||||
|
|
||||||
|
```{r}
|
||||||
|
afd_high_frequent %>% mutate(Worte = str_to_lower(Worte)) %>% inner_join(hitlerwords) -> afd_hitler_comparison
|
||||||
|
afdundfraktionslos_high_frequent %>% mutate(Worte = str_to_lower(Worte)) %>% inner_join(hitlerwords) -> afdundfraktionslos_hitler_comparison
|
||||||
|
grüne_high_frequent %>% mutate(Worte = str_to_lower(Worte)) %>% inner_join(hitlerwords) -> grüne_hitler_comparison
|
||||||
|
cdu_high_frequent %>% mutate(Worte = str_to_lower(Worte)) %>% inner_join(hitlerwords) -> cdu_hitler_comparison
|
||||||
|
linke_high_frequent %>% mutate(Worte = str_to_lower(Worte)) %>% inner_join(hitlerwords) -> linke_hitler_comparison
|
||||||
|
fdp_high_frequent %>% mutate(Worte = str_to_lower(Worte)) %>% inner_join(hitlerwords) -> fdp_hitler_comparison
|
||||||
|
fraktionslos_high_frequent %>% mutate(Worte = str_to_lower(Worte)) %>% inner_join(hitlerwords) -> fraktionslos_hitler_comparison
|
||||||
|
spd_high_frequent %>% mutate(Worte = str_to_lower(Worte)) %>% inner_join(hitlerwords) -> spd_hitler_comparison
|
||||||
|
na_high_frequent %>% mutate(Worte = str_to_lower(Worte)) %>% inner_join(hitlerwords) -> na_hitler_comparison
|
||||||
|
|
||||||
|
#not unique
|
||||||
|
tibble(fraction = c("AfD", "AfD&Fraktionslos", "BÜNDNIS 90 / DIE GRÜNEN", "CDU/CSU", "DIE LINKE", "FDP", "Fraktionslos", "SPD"),
|
||||||
|
absolute = c(nrow(afd_hitler_comparison), nrow(afdundfraktionslos_hitler_comparison), nrow(grüne_hitler_comparison), nrow(cdu_hitler_comparison), nrow(linke_hitler_comparison), nrow(fdp_hitler_comparison), nrow(fraktionslos_hitler_comparison), nrow(spd_hitler_comparison)),
|
||||||
|
total = c(nrow(afd_words), nrow(afdundfraktionslos_words), nrow(grüne_words), nrow(cdu_words), nrow(linke_words), nrow(fdp_words), nrow(fraktionslos_words), nrow(spd_words))
|
||||||
|
) %>% mutate(percent = 100*absolute/total) -> hitler_comparison
|
||||||
|
hitler_comparison
|
||||||
|
```
|
||||||
|
Finally, we want to plot our results:
|
||||||
|
```{r, fig.width=7}
|
||||||
|
bar_plot_fractions(hitler_comparison, y_variable = percent, title="Coincidence of party vocabulary with nazi vocabulary", ylab="unique 'nazi' words per total (unique) fraction words [%]")
|
||||||
|
```
|
||||||
@@ -0,0 +1,107 @@
|
|||||||
|
---
|
||||||
|
title: "Interaction between fractions"
|
||||||
|
output: rmarkdown::html_vignette
|
||||||
|
vignette: >
|
||||||
|
%\VignetteIndexEntry{Interaction between fractions}
|
||||||
|
%\VignetteEngine{knitr::rmarkdown}
|
||||||
|
%\VignetteEncoding{UTF-8}
|
||||||
|
---
|
||||||
|
|
||||||
|
```{r, include = FALSE}
|
||||||
|
knitr::opts_chunk$set(
|
||||||
|
collapse = TRUE,
|
||||||
|
comment = "#>"
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
```{r setup}
|
||||||
|
library(hateimparlament)
|
||||||
|
library(dplyr)
|
||||||
|
library(ggplot2)
|
||||||
|
library(stringr)
|
||||||
|
library(tidyr)
|
||||||
|
```
|
||||||
|
|
||||||
|
## Preparation of data
|
||||||
|
|
||||||
|
First, you need to download all records of the current legislative period.
|
||||||
|
```r
|
||||||
|
fetch_all("../inst/records/") # path to directory where records should be stored
|
||||||
|
```
|
||||||
|
Second, those `.xml` files, need to be parsed into `R` `tibbles`. This is accomplished by:
|
||||||
|
```r
|
||||||
|
read_all("../inst/records/") %>% repair() -> res
|
||||||
|
```
|
||||||
|
We also used `repair` to fix a bunch of formatting issues in the records.
|
||||||
|
|
||||||
|
For development purposes, we load the tables from csv files.
|
||||||
|
```{r}
|
||||||
|
res <- read_from_csv('../inst/csv/')
|
||||||
|
```
|
||||||
|
|
||||||
|
## Analysis
|
||||||
|
|
||||||
|
Now we can start analysing our parsed dataset:
|
||||||
|
|
||||||
|
### Which party gives the most applause to which parties?
|
||||||
|
|
||||||
|
```{r}
|
||||||
|
res$applause %>%
|
||||||
|
left_join(res$speaker, by=c("on_speaker" = "id")) %>%
|
||||||
|
select(on_fraction = fraction, where(is.logical)) %>%
|
||||||
|
group_by(on_fraction) %>%
|
||||||
|
arrange(on_fraction) %>%
|
||||||
|
summarize("AfD" = sum(`AfD`),
|
||||||
|
"BÜNDNIS 90/DIE GRÜNEN" = sum(`BUENDNIS_90_DIE_GRUENEN`),
|
||||||
|
"CDU/CSU" = sum(`CDU_CSU`),
|
||||||
|
"DIE LINKE" = sum(`DIE_LINKE`),
|
||||||
|
"FDP" = sum(`FDP`),
|
||||||
|
"SPD" = sum(`SPD`)) -> tb
|
||||||
|
```
|
||||||
|
|
||||||
|
For plotting our results we reorganize them a bit and produce a bar plot:
|
||||||
|
|
||||||
|
```{r, fig.width=7, fig.height=6}
|
||||||
|
pivot_longer(tb, where(is.numeric), "by_fraction", "count") %>%
|
||||||
|
filter(!is.na(on_fraction)) %>%
|
||||||
|
bar_plot_fractions(x_variable = on_fraction,
|
||||||
|
y_variable = value,
|
||||||
|
fill = by_fraction,
|
||||||
|
title = "Number of rounds of applauses from fractions to fractions",
|
||||||
|
xlab = "Applauded fraction",
|
||||||
|
ylab = "Rounds of applauses",
|
||||||
|
filllab = "Applauding fraction",
|
||||||
|
flipped = FALSE,
|
||||||
|
rotatelab = TRUE)
|
||||||
|
```
|
||||||
|
|
||||||
|
|
||||||
|
### Which party comments the most on which parties?
|
||||||
|
|
||||||
|
```{r}
|
||||||
|
res$comments %>%
|
||||||
|
left_join(res$speaker, by=c("on_speaker" = "id")) %>%
|
||||||
|
select(by_fraction = fraction.x, on_fraction = fraction.y) %>%
|
||||||
|
group_by(on_fraction) %>%
|
||||||
|
summarize(`AfD` = sum(str_detect(by_fraction, "AfD"), na.rm=T),
|
||||||
|
`BÜNDNIS 90/DIE GRÜNEN` = sum(str_detect(by_fraction, "BÜNDNIS 90/DIE GRÜNEN"), na.rm=T),
|
||||||
|
`CDU/CSU` = sum(str_detect(by_fraction, "CDU/CSU"), na.rm = T),
|
||||||
|
`DIE LINKE` = sum(str_detect(by_fraction, "DIE LINKE"), na.rm=T),
|
||||||
|
`FDP` = sum(str_detect(by_fraction, "FDP"), na.rm=T),
|
||||||
|
`SPD` = sum(str_detect(by_fraction, "SPD"), na.rm=T)) -> tb
|
||||||
|
```
|
||||||
|
Analogously we plot the results:
|
||||||
|
|
||||||
|
```{r, fig.width=7, fig.height=6}
|
||||||
|
pivot_longer(tb, where(is.numeric), "by_fraction", "count") %>%
|
||||||
|
filter(!is.na(on_fraction)) %>%
|
||||||
|
bar_plot_fractions(x_variable = on_fraction,
|
||||||
|
y_variable = value,
|
||||||
|
fill = by_fraction,
|
||||||
|
title = "Number of comments from fractions to fractions",
|
||||||
|
xlab = "Commented fraction",
|
||||||
|
ylab = "Number of comments",
|
||||||
|
filllab = "Commenting fraction",
|
||||||
|
flipped = FALSE,
|
||||||
|
rotatelab = TRUE)
|
||||||
|
```
|
||||||
Reference in New Issue
Block a user