EMZETT.
Login

Kurz: csv_text <- “name;punkte\nZoe;200\nBen;50\n” read.csv2(text = csv_text) read.table(text = “x y\n1 2\n3 4”, header = TRUE)

Teil des Kurses R

Kapitel 6 von 9 im Kurs R (Abschnitt „Praxis“). Mit Fortschritt, Quiz und Zertifikat auf der Lernseite.

Text bearbeiten

s <- "Hallo Welt, hallo R"
nchar(s)
toupper(s); tolower(s)
substr(s, 1, 5)
strsplit(s, ", ")[[1]]
strsplit("a,b;c", "[,;]")[[1]]
paste("a", "b", sep = "-"); paste0("x", 1:3); paste(c("a", "b"), collapse = "+")
sub("hallo", "Hi", s, ignore.case = TRUE)
gsub("l", "L", s)
grepl("Welt", s); grep("a", c("abc", "xyz", "bar")); grep("a", c("abc", "xyz", "bar"), value = TRUE)
regmatches(s, regexpr("[A-Z]\\w+", s))
regmatches("a1b22c333", gregexpr("[0-9]+", "a1b22c333"))[[1]]
sprintf("%s ist %d Jahre alt (%.1f m)", "Mia", 17L, 1.68)
sprintf("%05.1f|%-6s|%6s|%e", 3.14159, "ab", "cd", 12345.678)
format(1234567.891, big.mark = ".", decimal.mark = ",", nsmall = 2)
format(Sys.Date(), "%Y") >= "2025"
trimws("  getrimmt  "); rev(strsplit("abc", "")[[1]]); startsWith("Hallo", "Ha"); endsWith("Hallo", "lo")
toupper(letters[1:5]); LETTERS[24:26]; month.name[1:2]
sprintf("%s", c("a", "b")); nchar(c("eins", "zwei", "drei")); sort(c("b", "A", "c"))
casefold("ABC", upper = FALSE); chartr("abc", "xyz", "aabbcc"); rev(utf8ToInt("A")); intToUtf8(c(72, 105))

Ausgabe:

[1] 19
[1] "HALLO WELT, HALLO R"
[1] "hallo welt, hallo r"
[1] "Hallo"
[1] "Hallo Welt" "hallo R"
[1] "a" "b" "c"
[1] "a-b"
[1] "x1" "x2" "x3"
[1] "a+b"
[1] "Hi Welt, hallo R"
[1] "HaLLo WeLt, haLLo R"
[1] TRUE
[1] 1 3
[1] "abc" "bar"
[1] "Hallo"
[1] "1"   "22"  "333"
[1] "Mia ist 17 Jahre alt (1.7 m)"
[1] "003.1|ab    |    cd|1.234568e+04"
[1] "1.234.567,89"
[1] TRUE
[1] "getrimmt"
[1] "c" "b" "a"
[1] TRUE
[1] TRUE
[1] "A" "B" "C" "D" "E"
[1] "X" "Y" "Z"
[1] "January"  "February"
[1] "a" "b"
[1] 4 4 4
[1] "A" "b" "c"
[1] "abc"
[1] "xxyyzz"
[1] 65
[1] "Hi"

Reguläre Ausdrücke

text <- c("mia@example.com", "kein-email", "tom@test.de")
grepl("^[[:alnum:]._-]+@[[:alnum:].-]+\\.[a-z]{2,}$", text)
sub("@.*", "", text)
regexpr("@", text)
gsub("(\\w+)@(\\w+)", "\\2 at \\1", text[1])
gsub("\\s+", " ", "viel    Platz   hier")
sapply(strsplit(c("a b c", "d e"), " "), length)
unlist(regmatches("Preise: 10 Euro, 25 Euro", gregexpr("[0-9]+", "Preise: 10 Euro, 25 Euro")))

Ausgabe:

[1]  TRUE FALSE  TRUE
[1] "mia"        "kein-email" "tom"
[1]  4 -1  4
attr(,"match.length")
[1]  1 -1  1
attr(,"index.type")
[1] "chars"
attr(,"useBytes")
[1] TRUE
[1] "example at mia.com"
[1] "viel Platz hier"
[1] 3 2
[1] "10" "25"

Datum und Zeit

d <- as.Date("2025-03-15")
d + 20
format(d, "%d.%m.%Y"); format(d, "%A"); format(d, "%B"); weekdays(d); months(d)
as.numeric(as.Date("2025-12-24") - d)
difftime(as.Date("2025-12-24"), d, units = "weeks")
seq(d, by = "month", length.out = 4)
seq(d, as.Date("2025-03-18"), by = "day")
d > as.Date("2025-01-01")
as.Date("15.03.2025", format = "%d.%m.%Y")
format(as.Date("2024-02-29") + 365, "%Y-%m-%d")
cut(as.Date(c("2025-01-10", "2025-02-20", "2025-02-25")), "month")
t <- as.POSIXct("2025-03-15 14:30:00", tz = "UTC")
format(t, "%H:%M"); t + 3600; as.numeric(t)
trunc(t, "hours")
ISOdate(2025, 3, 15)
unclass(as.Date("1970-01-10"))

Ausgabe:

[1] "2025-04-04"
[1] "15.03.2025"
[1] "Saturday"
[1] "March"
[1] "Saturday"
[1] "March"
[1] 284
Time difference of 40.57143 weeks
[1] "2025-03-15" "2025-04-15" "2025-05-15" "2025-06-15"
[1] "2025-03-15" "2025-03-16" "2025-03-17" "2025-03-18"
[1] TRUE
[1] "2025-03-15"
[1] "2025-02-28"
[1] 2025-01-01 2025-02-01 2025-02-01
Levels: 2025-01-01 2025-02-01
[1] "14:30"
[1] "2025-03-15 15:30:00 UTC"
[1] 1742049000
[1] "2025-03-15 14:00:00 UTC"
[1] "2025-03-15 12:00:00 GMT"
[1] 9

Dateien lesen und schreiben

dat <- tempfile(fileext = ".csv")
df <- data.frame(name = c("Mia", "Tom"), punkte = c(120, 80))
write.csv(df, dat, row.names = FALSE)
cat(readLines(dat), sep = "\n")
gelesen <- read.csv(dat)
str(gelesen)
gelesen$punkte * 2
 
csv_text <- "name;punkte\nZoe;200\nBen;50\n"
read.csv2(text = csv_text)
read.table(text = "x y\n1 2\n3 4", header = TRUE)
 
txt <- tempfile()
writeLines(c("Zeile 1", "Zeile 2", "Zeile 3"), txt)
length(readLines(txt))
readLines(txt, n = 1)
cat("Anhängen\n", file = txt, append = TRUE)
length(readLines(txt))
file.exists(txt); file.remove(txt); file.exists(txt)
 
saveRDS(df, f <- tempfile(fileext = ".rds"))
identical(readRDS(f), df)
basename("/a/b/c.txt"); dirname("/a/b/c.txt"); tools::file_ext("daten.csv"); file.path("a", "b", "c.txt")

Ausgabe:

"name","punkte"
"Mia",120
"Tom",80
'data.frame':	2 obs. of  2 variables:
 $ name  : chr  "Mia" "Tom"
 $ punkte: int  120 80
[1] 240 160
  name punkte
1  Zoe    200
2  Ben     50
  x y
1 1 2
2 3 4
[1] 3
[1] "Zeile 1"
[1] 4
[1] TRUE
[1] TRUE
[1] FALSE
[1] TRUE
[1] "c.txt"
[1] "/a/b"
[1] "csv"
[1] "a/b/c.txt"

JSON und weitere Formate

Mit zusätzlichen Paketen: jsonlite (JSON), readxl (Excel), haven (SPSS/Stata), DBI (Datenbanken), arrow (Parquet):

install.packages(c("jsonlite", "readxl"))
library(jsonlite)
toJSON(list(name = "Mia", alter = 17), auto_unbox = TRUE)
fromJSON('{"a": 1, "b": [1, 2, 3]}')
readxl::read_excel("tabelle.xlsx", sheet = 1)

Merke

  • Text: paste, sprintf, substr, strsplit, gsub, grepl, nchar
  • Reguläre Ausdrücke in grepl, sub, gsub, regmatches
  • Datum: as.Date, format, difftime, seq(by = "month"), POSIXct für Zeiten
  • Dateien: read.csv, write.csv, readLines, writeLines, saveRDS
  • Mit Paketen: JSON, Excel, Datenbanken

Übungsaufgabe

Lies eine CSV mit Datum und Wert ein, wandle die Datumsspalte um und berechne den Monatsdurchschnitt.

Quiz zur Selbstkontrolle

Weiter im Kurs

Zurück: Kontrollfluss und Funktionen

Weiter: Statistik, Tests und Modelle

Alle Kapitel: R im Überblick