Text bearbeiten
s <- "Hallo Welt, hallo R"
nchar(s)
toupper(s); tolower(s)
substr(s, 1, 5)
strsplit(s, ", ")[[1]]
strsplit("a,b;c", "[,;]")[[1]]
paste("a", "b", sep = "-"); paste0("x", 1:3); paste(c("a", "b"), collapse = "+")
sub("hallo", "Hi", s, ignore.case = TRUE)
gsub("l", "L", s)
grepl("Welt", s); grep("a", c("abc", "xyz", "bar")); grep("a", c("abc", "xyz", "bar"), value = TRUE)
regmatches(s, regexpr("[A-Z]\\w+", s))
regmatches("a1b22c333", gregexpr("[0-9]+", "a1b22c333"))[[1]]
sprintf("%s ist %d Jahre alt (%.1f m)", "Mia", 17L, 1.68)
sprintf("%05.1f|%-6s|%6s|%e", 3.14159, "ab", "cd", 12345.678)
format(1234567.891, big.mark = ".", decimal.mark = ",", nsmall = 2)
format(Sys.Date(), "%Y") >= "2025"
trimws(" getrimmt "); rev(strsplit("abc", "")[[1]]); startsWith("Hallo", "Ha"); endsWith("Hallo", "lo")
toupper(letters[1:5]); LETTERS[24:26]; month.name[1:2]
sprintf("%s", c("a", "b")); nchar(c("eins", "zwei", "drei")); sort(c("b", "A", "c"))
casefold("ABC", upper = FALSE); chartr("abc", "xyz", "aabbcc"); rev(utf8ToInt("A")); intToUtf8(c(72, 105))[1] 19 [1] "HALLO WELT, HALLO R" [1] "hallo welt, hallo r" [1] "Hallo" [1] "Hallo Welt" "hallo R" [1] "a" "b" "c" [1] "a-b" [1] "x1" "x2" "x3" [1] "a+b" [1] "Hi Welt, hallo R" [1] "HaLLo WeLt, haLLo R" [1] TRUE [1] 1 3 [1] "abc" "bar" [1] "Hallo" [1] "1" "22" "333" [1] "Mia ist 17 Jahre alt (1.7 m)" [1] "003.1|ab | cd|1.234568e+04" [1] "1.234.567,89" [1] TRUE [1] "getrimmt" [1] "c" "b" "a" [1] TRUE [1] TRUE [1] "A" "B" "C" "D" "E" [1] "X" "Y" "Z" [1] "January" "February" [1] "a" "b" [1] 4 4 4 [1] "A" "b" "c" [1] "abc" [1] "xxyyzz" [1] 65 [1] "Hi"
Reguläre Ausdrücke
text <- c("mia@example.com", "kein-email", "tom@test.de")
grepl("^[[:alnum:]._-]+@[[:alnum:].-]+\\.[a-z]{2,}$", text)
sub("@.*", "", text)
regexpr("@", text)
gsub("(\\w+)@(\\w+)", "\\2 at \\1", text[1])
gsub("\\s+", " ", "viel Platz hier")
sapply(strsplit(c("a b c", "d e"), " "), length)
unlist(regmatches("Preise: 10 Euro, 25 Euro", gregexpr("[0-9]+", "Preise: 10 Euro, 25 Euro")))[1] TRUE FALSE TRUE [1] "mia" "kein-email" "tom" [1] 4 -1 4 attr(,"match.length") [1] 1 -1 1 attr(,"index.type") [1] "chars" attr(,"useBytes") [1] TRUE [1] "example at mia.com" [1] "viel Platz hier" [1] 3 2 [1] "10" "25"
Datum und Zeit
d <- as.Date("2025-03-15")
d + 20
format(d, "%d.%m.%Y"); format(d, "%A"); format(d, "%B"); weekdays(d); months(d)
as.numeric(as.Date("2025-12-24") - d)
difftime(as.Date("2025-12-24"), d, units = "weeks")
seq(d, by = "month", length.out = 4)
seq(d, as.Date("2025-03-18"), by = "day")
d > as.Date("2025-01-01")
as.Date("15.03.2025", format = "%d.%m.%Y")
format(as.Date("2024-02-29") + 365, "%Y-%m-%d")
cut(as.Date(c("2025-01-10", "2025-02-20", "2025-02-25")), "month")
t <- as.POSIXct("2025-03-15 14:30:00", tz = "UTC")
format(t, "%H:%M"); t + 3600; as.numeric(t)
trunc(t, "hours")
ISOdate(2025, 3, 15)
unclass(as.Date("1970-01-10"))[1] "2025-04-04" [1] "15.03.2025" [1] "Saturday" [1] "March" [1] "Saturday" [1] "March" [1] 284 Time difference of 40.57143 weeks [1] "2025-03-15" "2025-04-15" "2025-05-15" "2025-06-15" [1] "2025-03-15" "2025-03-16" "2025-03-17" "2025-03-18" [1] TRUE [1] "2025-03-15" [1] "2025-02-28" [1] 2025-01-01 2025-02-01 2025-02-01 Levels: 2025-01-01 2025-02-01 [1] "14:30" [1] "2025-03-15 15:30:00 UTC" [1] 1742049000 [1] "2025-03-15 14:00:00 UTC" [1] "2025-03-15 12:00:00 GMT" [1] 9
Dateien lesen und schreiben
dat <- tempfile(fileext = ".csv")
df <- data.frame(name = c("Mia", "Tom"), punkte = c(120, 80))
write.csv(df, dat, row.names = FALSE)
cat(readLines(dat), sep = "\n")
gelesen <- read.csv(dat)
str(gelesen)
gelesen$punkte * 2
csv_text <- "name;punkte\nZoe;200\nBen;50\n"
read.csv2(text = csv_text)
read.table(text = "x y\n1 2\n3 4", header = TRUE)
txt <- tempfile()
writeLines(c("Zeile 1", "Zeile 2", "Zeile 3"), txt)
length(readLines(txt))
readLines(txt, n = 1)
cat("Anhängen\n", file = txt, append = TRUE)
length(readLines(txt))
file.exists(txt); file.remove(txt); file.exists(txt)
saveRDS(df, f <- tempfile(fileext = ".rds"))
identical(readRDS(f), df)
basename("/a/b/c.txt"); dirname("/a/b/c.txt"); tools::file_ext("daten.csv"); file.path("a", "b", "c.txt")"name","punkte" "Mia",120 "Tom",80 'data.frame': 2 obs. of 2 variables: $ name : chr "Mia" "Tom" $ punkte: int 120 80 [1] 240 160 name punkte 1 Zoe 200 2 Ben 50 x y 1 1 2 2 3 4 [1] 3 [1] "Zeile 1" [1] 4 [1] TRUE [1] TRUE [1] FALSE [1] TRUE [1] "c.txt" [1] "/a/b" [1] "csv" [1] "a/b/c.txt"
JSON und weitere Formate
Mit zusätzlichen Paketen: jsonlite (JSON), readxl (Excel), haven (SPSS/Stata), DBI (Datenbanken), arrow (Parquet):
install.packages(c("jsonlite", "readxl"))
library(jsonlite)
toJSON(list(name = "Mia", alter = 17), auto_unbox = TRUE)
fromJSON('{"a": 1, "b": [1, 2, 3]}')
readxl::read_excel("tabelle.xlsx", sheet = 1)Merke
- Text:
paste,sprintf,substr,strsplit,gsub,grepl,nchar - Reguläre Ausdrücke in
grepl,sub,gsub,regmatches - Datum:
as.Date,format,difftime,seq(by = "month"),POSIXctfür Zeiten - Dateien:
read.csv,write.csv,readLines,writeLines,saveRDS - Mit Paketen: JSON, Excel, Datenbanken
Aufgabe
Lies eine CSV mit Datum und Wert ein, wandle die Datumsspalte um und berechne den Monatsdurchschnitt.