blob: a68f03eb62cb834a13b2a7702efaf3cbd56925a8 [file] [log] [blame]
Marc Kupietza8c40f42025-06-24 15:49:52 +02001#' KorAPQuery class (internal)
Marc Kupietze95108e2019-09-18 13:23:58 +02002#'
Marc Kupietza8c40f42025-06-24 15:49:52 +02003#' Internal class for query state management. Users work with `corpusQuery()`, `fetchAll()`, and `fetchNext()` instead.
Marc Kupietze95108e2019-09-18 13:23:58 +02004#'
Marc Kupietza8c40f42025-06-24 15:49:52 +02005#' @keywords internal
Marc Kupietze95108e2019-09-18 13:23:58 +02006#' @include KorAPConnection.R
Marc Kupietz6dfeed92025-06-03 11:58:06 +02007#' @include logging.R
Marc Kupietzf9129592025-01-26 19:17:54 +01008#' @import httr2
Marc Kupietze95108e2019-09-18 13:23:58 +02009#'
Marc Kupietza6e4ee62021-03-05 09:00:15 +010010#' @include RKorAPClient-package.R
Marc Kupietz5bbc9db2019-08-30 16:30:45 +020011
Marc Kupietze95108e2019-09-18 13:23:58 +020012#' @export
13KorAPQuery <- setClass("KorAPQuery", slots = c(
Marc Kupietzb8972182019-09-20 21:33:46 +020014 "korapConnection",
Marc Kupietze95108e2019-09-18 13:23:58 +020015 "request",
16 "vc",
17 "totalResults",
18 "nextStartIndex",
19 "fields",
20 "requestUrl",
21 "webUIRequestUrl",
22 "apiResponse",
23 "collectedMatches",
Marc Kupietza29f3d42025-07-18 10:14:43 +020024 "hasMoreMatches"
Marc Kupietze95108e2019-09-18 13:23:58 +020025))
Marc Kupietz5bbc9db2019-08-30 16:30:45 +020026
Marc Kupietza8c40f42025-06-24 15:49:52 +020027#' Initialize KorAPQuery object
28#' @keywords internal
Marc Kupietze95108e2019-09-18 13:23:58 +020029#' @param .Object …
Marc Kupietzb8972182019-09-20 21:33:46 +020030#' @param korapConnection KorAPConnection object
Marc Kupietze95108e2019-09-18 13:23:58 +020031#' @param request query part of the request URL
32#' @param vc definition of a virtual corpus
33#' @param totalResults number of hits the query has yielded
34#' @param nextStartIndex at what index to start the next fetch of query results
35#' @param fields what data / metadata fields should be collected
36#' @param requestUrl complete URL of the API request
37#' @param webUIRequestUrl URL of a web frontend request corresponding to the API request
38#' @param apiResponse data-frame representation of the JSON response of the API request
Marc Kupietz7776dec2019-09-27 16:59:02 +020039#' @param hasMoreMatches logical that signals if more query results can be fetched
Marc Kupietze95108e2019-09-18 13:23:58 +020040#' @param collectedMatches matches already fetched from the KorAP-API-server
Marc Kupietz97a1bca2019-10-04 22:52:09 +020041#'
42#' @importFrom tibble tibble
Marc Kupietze95108e2019-09-18 13:23:58 +020043#' @export
Marc Kupietzd8851222025-05-01 10:57:19 +020044setMethod(
45 "initialize", "KorAPQuery",
46 function(.Object, korapConnection = NULL, request = NULL, vc = "", totalResults = 0, nextStartIndex = 0, fields = c(
47 "corpusSigle", "textSigle", "pubDate", "pubPlace",
48 "availability", "textClass", "snippet", "tokens"
49 ),
Marc Kupietza29f3d42025-07-18 10:14:43 +020050 requestUrl = "", webUIRequestUrl = "", apiResponse = NULL, hasMoreMatches = FALSE, collectedMatches = NULL) {
Marc Kupietzd8851222025-05-01 10:57:19 +020051 .Object <- callNextMethod()
52 .Object@korapConnection <- korapConnection
53 .Object@request <- request
54 .Object@vc <- vc
55 .Object@totalResults <- totalResults
56 .Object@nextStartIndex <- nextStartIndex
57 .Object@fields <- fields
58 .Object@requestUrl <- requestUrl
59 .Object@webUIRequestUrl <- webUIRequestUrl
60 .Object@apiResponse <- apiResponse
61 .Object@hasMoreMatches <- hasMoreMatches
62 .Object@collectedMatches <- collectedMatches
63 .Object
64 }
65)
Marc Kupietz632cbd42019-09-06 16:04:51 +020066
Marc Kupietzd8851222025-05-01 10:57:19 +020067setGeneric("corpusQuery", function(kco, ...) standardGeneric("corpusQuery"))
68setGeneric("fetchAll", function(kqo, ...) standardGeneric("fetchAll"))
69setGeneric("fetchNext", function(kqo, ...) standardGeneric("fetchNext"))
70setGeneric("fetchRest", function(kqo, ...) standardGeneric("fetchRest"))
Marc Kupietz0af75932025-09-09 18:14:16 +020071setGeneric(
72 "fetchAnnotations",
73 function(kqo,
74 foundry = "tt",
75 overwrite = FALSE,
76 verbose = kqo@korapConnection@verbose) standardGeneric("fetchAnnotations")
77)
Marc Kupietzd8851222025-05-01 10:57:19 +020078setGeneric("frequencyQuery", function(kco, ...) standardGeneric("frequencyQuery"))
Marc Kupietze95108e2019-09-18 13:23:58 +020079
80maxResultsPerPage <- 50
Marc Kupietz62da2b52019-09-12 17:43:34 +020081
Marc Kupietz4de53ec2019-10-04 09:12:00 +020082## quiets concerns of R CMD check re: the .'s that appear in pipelines
Marc Kupietzef1ef4a2025-02-19 12:12:40 +010083utils::globalVariables(c("."))
Marc Kupietz632cbd42019-09-06 16:04:51 +020084
Marc Kupietza8c40f42025-06-24 15:49:52 +020085#' Search corpus for query terms
Marc Kupietzdbd431a2021-08-29 12:17:45 +020086#'
Marc Kupietz67edcb52021-09-20 21:54:24 +020087#' **`corpusQuery`** performs a corpus query via a connection to a KorAP-API-server
Marc Kupietze95108e2019-09-18 13:23:58 +020088#'
Marc Kupietza8c40f42025-06-24 15:49:52 +020089#' @family corpus search functions
Marc Kupietzdbd431a2021-08-29 12:17:45 +020090#' @aliases corpusQuery
91#'
92#' @importFrom urltools url_encode
93#' @importFrom purrr pmap
Marc Kupietzea34b812025-06-25 15:49:00 +020094#' @importFrom dplyr bind_rows group_by
Marc Kupietzdbd431a2021-08-29 12:17:45 +020095#'
Marc Kupietz617266d2025-02-27 10:43:07 +010096#' @param kco [KorAPConnection()] object (obtained e.g. from `KorAPConnection()`
Marc Kupietz67edcb52021-09-20 21:54:24 +020097#' @param query string that contains the corpus query. The query language depends on the `ql` parameter. Either `query` must be provided or `KorAPUrl`.
Marc Kupietz632cbd42019-09-06 16:04:51 +020098#' @param vc string describing the virtual corpus in which the query should be performed. An empty string (default) means the whole corpus, as far as it is license-wise accessible.
Marc Kupietz67edcb52021-09-20 21:54:24 +020099#' @param KorAPUrl instead of providing the query and vc string parameters, you can also simply copy a KorAP query URL from your browser and use it here (and in `KorAPConnection`) to provide all necessary information for the query.
Marc Kupietz132f0052023-04-16 14:23:05 +0200100#' @param metadataOnly logical that determines whether queries should return only metadata without any snippets. This can also be useful to prevent access rewrites. Note that the default value is TRUE.
101#' If you want your corpus queries to return not only metadata, but also KWICS, you need to authorize
102#' your RKorAPClient application as explained in the
103#' [authorization section](https://github.com/KorAP/RKorAPClient#authorization)
104#' of the RKorAPClient Readme on GitHub and set the `metadataOnly` parameter to
105#' `FALSE`.
Marc Kupietz67edcb52021-09-20 21:54:24 +0200106#' @param ql string to choose the query language (see [section on Query Parameters](https://github.com/KorAP/Kustvakt/wiki/Service:-Search-GET#user-content-parameters) in the Kustvakt-Wiki for possible values.
Marc Kupietz1623fe82025-06-24 16:31:46 +0200107#' @param fields character vector specifying which metadata fields to retrieve for each match.
108#' Available fields depend on the corpus. For DeReKo (German Reference Corpus), possible fields include:
109#' \describe{
110#' \item{**Text identification**:}{`textSigle`, `docSigle`, `corpusSigle` - hierarchical text identifiers}
111#' \item{**Publication info**:}{`author`, `editor`, `title`, `docTitle`, `corpusTitle` - authorship and titles}
112#' \item{**Temporal data**:}{`pubDate`, `creationDate` - when text was published/created}
113#' \item{**Publication details**:}{`pubPlace`, `publisher`, `reference` - where/how published}
114#' \item{**Text classification**:}{`textClass`, `textType`, `textTypeArt`, `textDomain`, `textColumn` - topic domain, genre, text type and column}
115#' \item{**Adminstrative and technical info**:}{`corpusEditor`, `availability`, `language`, `foundries` - access rights and annotations}
116#' \item{**Content data**:}{`snippet`, `tokens`, `tokenSource`, `externalLink` - actual text content, tokenization, and link to source text}
117#' \item{**System data**:}{`indexCreationDate`, `indexLastModified` - corpus indexing info}
118#' }
119#' Use `c("textSigle", "pubDate", "author")` to retrieve multiple fields.
120#' Default fields provide basic text identification and publication metadata. The actual text content (`snippet` and `tokens`) are activated by default if `metadataOnly` is set to `FALSE`.
Marc Kupietz43a6ade2020-02-18 17:01:44 +0100121#' @param accessRewriteFatal abort if query or given vc had to be rewritten due to insufficient rights (not yet implemented).
Marc Kupietz25aebc32019-09-16 18:40:50 +0200122#' @param verbose print some info
Marc Kupietz4de53ec2019-10-04 09:12:00 +0200123#' @param as.df return result as data frame instead of as S4 object?
Marc Kupietzad8d2ed2025-04-05 15:37:38 +0200124#' @param expand logical that decides if `query` and `vc` parameters are expanded to all of their combinations. Defaults to `TRUE`, iff `query` and `vc` have different lengths
Marc Kupietzd9b2fd72023-04-17 19:08:50 +0200125#' @param context string that specifies the size of the left and the right context returned in `snippet`
126#' (provided that `metadataOnly` is set to `false` and that the necessary access right are met).
127#' The format of the context size specifcation (e.g. `3-token,3-token`) is described in the [Service: Search GET documentation of the Kustvakt Wiki](https://github.com/KorAP/Kustvakt/wiki/Service:-Search-GET).
128#' If the parameter is not set, the default context size secification of the KorAP server instance will be used.
129#' Note that you cannot overrule the maximum context size set in the KorAP server instance,
130#' as this is typically legally motivated.
Marc Kupietzad8d2ed2025-04-05 15:37:38 +0200131#' @return Depending on the `as.df` parameter, a tibble or a [KorAPQuery()] object that, among other information, contains the total number of results in `@totalResults`. The resulting object can be used to fetch all query results (with [fetchAll()]) or the next page of results (with [fetchNext()]).
Marc Kupietz67edcb52021-09-20 21:54:24 +0200132#' A corresponding URL to be used within a web browser is contained in `@webUIRequestUrl`
133#' Please make sure to check `$collection$rewrites` to see if any unforeseen access rewrites of the query's virtual corpus had to be performed.
Marc Kupietz632cbd42019-09-06 16:04:51 +0200134#'
135#' @examples
Marc Kupietz6ae76052021-09-21 10:34:00 +0200136#' \dontrun{
137#'
Marc Kupietz1623fe82025-06-24 16:31:46 +0200138#' # Fetch basic metadata for "Ameisenplage"
Marc Kupietzd3526422025-06-25 09:16:15 +0200139#' KorAPConnection() |>
140#' corpusQuery("Ameisenplage") |>
Marc Kupietzd8851222025-05-01 10:57:19 +0200141#' fetchAll()
Marc Kupietz1623fe82025-06-24 16:31:46 +0200142#'
143#' # Fetch specific metadata fields for bibliographic analysis
Marc Kupietzd3526422025-06-25 09:16:15 +0200144#' query <- KorAPConnection() |>
Marc Kupietz1623fe82025-06-24 16:31:46 +0200145#' corpusQuery("Ameisenplage",
146#' fields = c("textSigle", "author", "title", "pubDate", "pubPlace", "textType"))
147#' results <- fetchAll(query)
148#' results@collectedMatches
Marc Kupietz657d8e72020-02-25 18:31:50 +0100149#' }
Marc Kupietz3c531f62019-09-13 12:17:24 +0200150#'
Marc Kupietz6ae76052021-09-21 10:34:00 +0200151#' \dontrun{
152#'
Marc Kupietz603491f2019-09-18 14:01:02 +0200153#' # Use the copy of a KorAP-web-frontend URL for an API query of "Ameise" in a virtual corpus
154#' # and show the number of query hits (but don't fetch them).
Marc Kupietz69cc54a2019-09-30 12:06:54 +0200155#'
Marc Kupietzd3526422025-06-25 09:16:15 +0200156#' KorAPConnection(verbose = TRUE) |>
Marc Kupietzd8851222025-05-01 10:57:19 +0200157#' corpusQuery(
158#' KorAPUrl =
159#' "https://korap.ids-mannheim.de/?q=Ameise&cq=pubDate+since+2017&ql=poliqarp"
160#' )
Marc Kupietz6ae76052021-09-21 10:34:00 +0200161#' }
162#'
163#' \dontrun{
Marc Kupietz3c531f62019-09-13 12:17:24 +0200164#'
Marc Kupietz603491f2019-09-18 14:01:02 +0200165#' # Plot the time/frequency curve of "Ameisenplage"
Marc Kupietzd3526422025-06-25 09:16:15 +0200166#' KorAPConnection(verbose = TRUE) |>
Marc Kupietzd8851222025-05-01 10:57:19 +0200167#' {
168#' . ->> kco
Marc Kupietzd3526422025-06-25 09:16:15 +0200169#' } |>
170#' corpusQuery("Ameisenplage") |>
171#' fetchAll() |>
172#' slot("collectedMatches") |>
173#' mutate(year = lubridate::year(pubDate)) |>
174#' dplyr::select(year) |>
175#' group_by(year) |>
176#' summarise(Count = dplyr::n()) |>
Marc Kupietzd8851222025-05-01 10:57:19 +0200177#' mutate(Freq = mapply(function(f, y) {
178#' f / corpusStats(kco, paste("pubDate in", y))@tokens
Marc Kupietzd3526422025-06-25 09:16:15 +0200179#' }, Count, year)) |>
180#' dplyr::select(-Count) |>
181#' complete(year = min(year):max(year), fill = list(Freq = 0)) |>
Marc Kupietz69cc54a2019-09-30 12:06:54 +0200182#' plot(type = "l")
Marc Kupietz05b22772020-02-18 21:58:42 +0100183#' }
Marc Kupietz67edcb52021-09-20 21:54:24 +0200184#' @seealso [KorAPConnection()], [fetchNext()], [fetchRest()], [fetchAll()], [corpusStats()]
Marc Kupietz632cbd42019-09-06 16:04:51 +0200185#'
186#' @references
Marc Kupietz67edcb52021-09-20 21:54:24 +0200187#' <https://ids-pub.bsz-bw.de/frontdoor/index/index/docId/9026>
Marc Kupietz632cbd42019-09-06 16:04:51 +0200188#'
189#' @export
Marc Kupietzd8851222025-05-01 10:57:19 +0200190setMethod(
191 "corpusQuery", "KorAPConnection",
192 function(kco,
193 query = if (missing(KorAPUrl)) {
194 stop("At least one of the parameters query and KorAPUrl must be specified.", call. = FALSE)
195 } else {
196 httr2::url_parse(KorAPUrl)$query$q
197 },
198 vc = if (missing(KorAPUrl)) "" else httr2::url_parse(KorAPUrl)$query$cq,
199 KorAPUrl,
200 metadataOnly = TRUE,
201 ql = if (missing(KorAPUrl)) "poliqarp" else httr2::url_parse(KorAPUrl)$query$ql,
202 fields = c(
203 "corpusSigle",
204 "textSigle",
205 "pubDate",
206 "pubPlace",
207 "availability",
208 "textClass",
209 "snippet",
210 "tokens"
211 ),
212 accessRewriteFatal = TRUE,
213 verbose = kco@verbose,
214 expand = length(vc) != length(query),
215 as.df = FALSE,
216 context = NULL) {
217 if (length(query) > 1 || length(vc) > 1) {
Marc Kupietzf632fe32026-09-08 07:58:46 +0200218 # expand_grid() and tibble() drop the names of vc, so the labels the
219 # caller gave their virtual corpora are carried along as a column
220 vcLabel <- vcLabels(vc)
Marc Kupietzd8851222025-05-01 10:57:19 +0200221 grid <- if (expand) expand_grid(query = query, vc = vc) else tibble(query = query, vc = vc)
Marc Kupietzf632fe32026-09-08 07:58:46 +0200222 if (!is.null(vcLabel)) {
223 grid$label <- if (expand) rep(vcLabel, times = length(query)) else vcLabel
224 }
Marc Kupietz6ef61a82025-05-29 16:07:03 +0200225
226 # Initialize timing variables for ETA calculation
227 total_queries <- nrow(grid)
228 current_query <- 0
229 start_time <- Sys.time()
230
Marc Kupietzf632fe32026-09-08 07:58:46 +0200231 results <- purrr::pmap(grid, function(query, vc, label = NULL, ...) {
Marc Kupietz6ef61a82025-05-29 16:07:03 +0200232 current_query <<- current_query + 1
233
234 # Execute the single query directly (avoiding recursive call)
235 contentFields <- c("snippet", "tokens")
236 query_fields <- fields
237 if (metadataOnly) {
238 query_fields <- query_fields[!query_fields %in% contentFields]
239 }
240 if (!"textSigle" %in% query_fields) {
241 query_fields <- c(query_fields, "textSigle")
242 }
243 request <-
244 paste0(
245 "?q=",
246 url_encode(enc2utf8(query)),
247 ifelse(!metadataOnly && !is.null(context) && context != "", paste0("&context=", url_encode(enc2utf8(context))), ""),
248 ifelse(vc != "", paste0("&cq=", url_encode(enc2utf8(vc))), ""),
249 ifelse(!metadataOnly, "&show-tokens=true", ""),
250 "&ql=", ql
251 )
252 webUIRequestUrl <- paste0(kco@KorAPUrl, request)
253 requestUrl <- paste0(
254 kco@apiUrl,
255 "search",
256 request,
257 "&fields=",
258 paste(query_fields, collapse = ","),
259 if (metadataOnly) "&access-rewrite-disabled=true" else ""
260 )
261
262 # Show individual query progress
263 log_info(verbose, "\rSearching \"", query, "\" in \"", vc, "\"", sep = "")
264 res <- apiCall(kco, paste0(requestUrl, "&count=0"))
265 if (is.null(res)) {
266 log_info(verbose, ": API call failed\n")
267 totalResults <- 0
268 } else {
269 totalResults <- as.integer(res$meta$totalResults)
270 log_info(verbose, ": ", totalResults, " hits")
271 if (!is.null(res$meta$cached)) {
272 log_info(verbose, " [cached]")
273 } else if (!is.null(res$meta$benchmark)) {
274 if (is.character(res$meta$benchmark) && grepl("s$", res$meta$benchmark)) {
275 time_value <- as.numeric(sub("s$", "", res$meta$benchmark))
276 formatted_time <- paste0(round(time_value, 2), "s")
277 log_info(verbose, ", took ", formatted_time)
278 } else {
279 log_info(verbose, ", took ", res$meta$benchmark)
280 }
281 }
Marc Kupietz365660e2025-06-25 15:09:55 +0200282
283 # Calculate and display ETA information on the same line if verbose and we have more than one query
284 if (verbose && total_queries > 1) {
285 eta_info <- calculate_eta(current_query, total_queries, start_time)
286 if (eta_info != "") {
287 elapsed_time <- as.numeric(difftime(Sys.time(), start_time, units = "secs"))
288 avg_time_per_query <- elapsed_time / current_query
289
290 # Add ETA info to the same line - remove the leading ". " for cleaner formatting
291 clean_eta_info <- sub("^\\. ", ". ", eta_info)
292 log_info(verbose, clean_eta_info)
293 }
294 }
295
Marc Kupietz6ef61a82025-05-29 16:07:03 +0200296 log_info(verbose, "\n")
297 }
298
299 result <- data.frame(
300 query = query,
301 totalResults = totalResults,
302 vc = vc,
303 webUIRequestUrl = webUIRequestUrl,
304 stringsAsFactors = FALSE
305 )
Marc Kupietzf632fe32026-09-08 07:58:46 +0200306 if (!is.null(label)) {
307 result <- tibble::add_column(result, label = label, .after = "vc")
308 }
Marc Kupietz6ef61a82025-05-29 16:07:03 +0200309
Marc Kupietz6ef61a82025-05-29 16:07:03 +0200310 return(result)
311 })
312
313 results %>% bind_rows()
Marc Kupietzd8851222025-05-01 10:57:19 +0200314 } else {
Marc Kupietz2078bde2023-08-27 16:46:15 +0200315 contentFields <- c("snippet", "tokens")
Marc Kupietza96537f2019-11-09 23:07:44 +0100316 if (metadataOnly) {
317 fields <- fields[!fields %in% contentFields]
318 }
Marc Kupietz80dc6432025-02-07 16:57:40 +0100319 if (!"textSigle" %in% fields) {
320 fields <- c(fields, "textSigle")
321 }
Marc Kupietza96537f2019-11-09 23:07:44 +0100322 request <-
Marc Kupietzd8851222025-05-01 10:57:19 +0200323 paste0(
324 "?q=",
325 url_encode(enc2utf8(query)),
326 ifelse(!metadataOnly && !is.null(context) && context != "", paste0("&context=", url_encode(enc2utf8(context))), ""),
327 ifelse(vc != "", paste0("&cq=", url_encode(enc2utf8(vc))), ""),
328 ifelse(!metadataOnly, "&show-tokens=true", ""),
329 "&ql=", ql
330 )
Marc Kupietza96537f2019-11-09 23:07:44 +0100331 webUIRequestUrl <- paste0(kco@KorAPUrl, request)
332 requestUrl <- paste0(
333 kco@apiUrl,
Marc Kupietzd8851222025-05-01 10:57:19 +0200334 "search",
Marc Kupietza96537f2019-11-09 23:07:44 +0100335 request,
Marc Kupietzd8851222025-05-01 10:57:19 +0200336 "&fields=",
Marc Kupietza96537f2019-11-09 23:07:44 +0100337 paste(fields, collapse = ","),
Marc Kupietzd8851222025-05-01 10:57:19 +0200338 if (metadataOnly) "&access-rewrite-disabled=true" else ""
Marc Kupietza96537f2019-11-09 23:07:44 +0100339 )
Marc Kupietzd8851222025-05-01 10:57:19 +0200340 log_info(verbose, "\rSearching \"", query, "\" in \"", vc, "\"",
341 sep =
342 ""
343 )
344 res <- apiCall(kco, paste0(requestUrl, "&count=0"))
Marc Kupietza4675722022-02-23 23:55:15 +0100345 if (is.null(res)) {
Marc Kupietza4675722022-02-23 23:55:15 +0100346 message("API call failed.")
347 totalResults <- 0
348 } else {
Marc Kupietzd8851222025-05-01 10:57:19 +0200349 totalResults <- as.integer(res$meta$totalResults)
Marc Kupietza47d1502023-04-18 15:26:47 +0200350 log_info(verbose, ": ", totalResults, " hits")
Marc Kupietzd8851222025-05-01 10:57:19 +0200351 if (!is.null(res$meta$cached)) {
Marc Kupietza47d1502023-04-18 15:26:47 +0200352 log_info(verbose, " [cached]\n")
Marc Kupietzd8851222025-05-01 10:57:19 +0200353 } else if (!is.null(res$meta$benchmark)) {
Marc Kupietz2baf5c52025-09-05 16:41:11 +0200354 # Round the benchmark time to 2 decimal places for better readability.
355 # Be robust to locales using comma as decimal separator (e.g., "0,12s").
Marc Kupietz7638ca42025-05-25 13:18:16 +0200356 if (is.character(res$meta$benchmark) && grepl("s$", res$meta$benchmark)) {
Marc Kupietz2baf5c52025-09-05 16:41:11 +0200357 bench_str <- sub("s$", "", res$meta$benchmark)
358 bench_num <- suppressWarnings(as.numeric(gsub(",", ".", bench_str)))
359 if (!is.na(bench_num)) {
360 formatted_time <- paste0(round(bench_num, 2), "s")
361 } else {
362 formatted_time <- res$meta$benchmark
363 }
Marc Kupietz7638ca42025-05-25 13:18:16 +0200364 log_info(verbose, ", took ", formatted_time, "\n", sep = "")
365 } else {
366 # Fallback if the format is different than expected
367 log_info(verbose, ", took ", res$meta$benchmark, "\n", sep = "")
368 }
Marc Kupietzd8851222025-05-01 10:57:19 +0200369 } else {
370 log_info(verbose, "\n")
371 }
Marc Kupietza4675722022-02-23 23:55:15 +0100372 }
Marc Kupietzd8851222025-05-01 10:57:19 +0200373 if (as.df) {
Marc Kupietza96537f2019-11-09 23:07:44 +0100374 data.frame(
375 query = query,
Marc Kupietza4675722022-02-23 23:55:15 +0100376 totalResults = totalResults,
Marc Kupietza96537f2019-11-09 23:07:44 +0100377 vc = vc,
378 webUIRequestUrl = webUIRequestUrl,
379 stringsAsFactors = FALSE
380 )
Marc Kupietzd8851222025-05-01 10:57:19 +0200381 } else {
Marc Kupietza96537f2019-11-09 23:07:44 +0100382 KorAPQuery(
383 korapConnection = kco,
384 nextStartIndex = 0,
385 fields = fields,
386 requestUrl = requestUrl,
387 request = request,
Marc Kupietza4675722022-02-23 23:55:15 +0100388 totalResults = totalResults,
Marc Kupietza96537f2019-11-09 23:07:44 +0100389 vc = vc,
390 apiResponse = res,
391 webUIRequestUrl = webUIRequestUrl,
Marc Kupietza4675722022-02-23 23:55:15 +0100392 hasMoreMatches = (totalResults > 0),
Marc Kupietza96537f2019-11-09 23:07:44 +0100393 )
Marc Kupietzd8851222025-05-01 10:57:19 +0200394 }
Marc Kupietza96537f2019-11-09 23:07:44 +0100395 }
Marc Kupietzd8851222025-05-01 10:57:19 +0200396 }
397)
Marc Kupietz5bbc9db2019-08-30 16:30:45 +0200398
Marc Kupietz05a60792024-12-07 16:23:31 +0100399#' @importFrom purrr map
400repair_data_strcuture <- function(x) {
Marc Kupietzd8851222025-05-01 10:57:19 +0200401 if (is.list(x)) {
402 as.character(purrr::map(x, ~ if (length(.x) > 1) {
Marc Kupietz05a60792024-12-07 16:23:31 +0100403 paste(.x, collapse = " ")
404 } else {
405 .x
406 }))
Marc Kupietzd8851222025-05-01 10:57:19 +0200407 } else {
Marc Kupietz05a60792024-12-07 16:23:31 +0100408 ifelse(is.na(x), "", x)
Marc Kupietzd8851222025-05-01 10:57:19 +0200409 }
Marc Kupietz05a60792024-12-07 16:23:31 +0100410}
411
Marc Kupietz62da2b52019-09-12 17:43:34 +0200412#' Fetch the next bunch of results of a KorAP query.
Marc Kupietze95108e2019-09-18 13:23:58 +0200413#'
Marc Kupietz67edcb52021-09-20 21:54:24 +0200414#' **`fetchNext`** fetches the next bunch of results of a KorAP query.
Marc Kupietz3f575282019-10-04 14:46:04 +0200415#'
Marc Kupietza8c40f42025-06-24 15:49:52 +0200416#' @family corpus search functions
417#'
Marc Kupietz67edcb52021-09-20 21:54:24 +0200418#' @param kqo object obtained from [corpusQuery()]
Marc Kupietz62da2b52019-09-12 17:43:34 +0200419#' @param offset start offset for query results to fetch
420#' @param maxFetch maximum number of query results to fetch
Marc Kupietz25aebc32019-09-16 18:40:50 +0200421#' @param verbose print progress information if true
Marc Kupietz67edcb52021-09-20 21:54:24 +0200422#' @param randomizePageOrder fetch result pages in pseudo random order if true. Use [set.seed()] to set seed for reproducible results.
423#' @return The `kqo` input object with updated slots `collectedMatches`, `apiResponse`, `nextStartIndex`, `hasMoreMatches`
Marc Kupietz62da2b52019-09-12 17:43:34 +0200424#'
Marc Kupietz05b22772020-02-18 21:58:42 +0100425#' @examples
Marc Kupietz6ae76052021-09-21 10:34:00 +0200426#' \dontrun{
427#'
Marc Kupietzd3526422025-06-25 09:16:15 +0200428#' q <- KorAPConnection() |>
429#' corpusQuery("Ameisenplage") |>
Marc Kupietzd8851222025-05-01 10:57:19 +0200430#' fetchNext()
Marc Kupietz05b22772020-02-18 21:58:42 +0100431#' q@collectedMatches
Marc Kupietz657d8e72020-02-25 18:31:50 +0100432#' }
Marc Kupietz05b22772020-02-18 21:58:42 +0100433#'
Marc Kupietz62da2b52019-09-12 17:43:34 +0200434#' @references
Marc Kupietz67edcb52021-09-20 21:54:24 +0200435#' <https://ids-pub.bsz-bw.de/frontdoor/index/index/docId/9026>
Marc Kupietz62da2b52019-09-12 17:43:34 +0200436#'
Marc Kupietze95108e2019-09-18 13:23:58 +0200437#' @aliases fetchNext
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200438#' @importFrom dplyr rowwise mutate bind_rows select summarise n select
Marc Kupietzf4881122024-12-17 14:55:39 +0100439#' @importFrom tibble enframe add_column
440#' @importFrom stringr word
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200441#' @importFrom tidyr unnest unchop pivot_wider
442#' @importFrom purrr map
Marc Kupietz632cbd42019-09-06 16:04:51 +0200443#' @export
Marc Kupietzdbd431a2021-08-29 12:17:45 +0200444setMethod("fetchNext", "KorAPQuery", function(kqo,
445 offset = kqo@nextStartIndex,
446 maxFetch = maxResultsPerPage,
447 verbose = kqo@korapConnection@verbose,
448 randomizePageOrder = FALSE) {
Marc Kupietza7a8f1b2024-12-18 15:56:19 +0100449 # https://stackoverflow.com/questions/8096313/no-visible-binding-for-global-variable-note-in-r-cmd-check
Marc Kupietzd8851222025-05-01 10:57:19 +0200450 results <- key <- name <- tmp_positions <- 0
Marc Kupietza7a8f1b2024-12-18 15:56:19 +0100451
Marc Kupietze95108e2019-09-18 13:23:58 +0200452 if (kqo@totalResults == 0 || offset >= kqo@totalResults) {
453 return(kqo)
Marc Kupietz62da2b52019-09-12 17:43:34 +0200454 }
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200455 use_korap_api <- Sys.getenv("USE_KORAP_API", unset = NA)
Marc Kupietz623d7122025-05-25 12:46:12 +0200456 # Calculate the initial page number (not used directly - keeping for reference)
Marc Kupietze95108e2019-09-18 13:23:58 +0200457 collectedMatches <- kqo@collectedMatches
Marc Kupietz62da2b52019-09-12 17:43:34 +0200458
Marc Kupietz24799fd2025-06-25 14:15:36 +0200459 # Track start time for ETA calculation
460 start_time <- Sys.time()
461
Marc Kupietz623d7122025-05-25 12:46:12 +0200462 # For randomized page order, generate a list of randomized page indices
Marc Kupietzdbd431a2021-08-29 12:17:45 +0200463 if (randomizePageOrder) {
Marc Kupietz623d7122025-05-25 12:46:12 +0200464 # Calculate how many pages we need to fetch based on maxFetch
465 total_pages_to_fetch <- if (!is.na(maxFetch)) {
466 # Either limited by maxFetch or total results, whichever is smaller
467 min(ceiling(maxFetch / maxResultsPerPage), ceiling(kqo@totalResults / maxResultsPerPage))
468 } else {
469 # All pages
470 ceiling(kqo@totalResults / maxResultsPerPage)
471 }
472
473 # Generate randomized page indices (0-based for API)
474 pages <- sample.int(ceiling(kqo@totalResults / maxResultsPerPage), total_pages_to_fetch) - 1
475 page_index <- 1 # Index to track which page in the randomized list we're on
Marc Kupietzdbd431a2021-08-29 12:17:45 +0200476 }
477
Marc Kupietzd8851222025-05-01 10:57:19 +0200478 if (is.null(collectedMatches)) {
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200479 collectedMatches <- data.frame()
480 }
Marc Kupietz623d7122025-05-25 12:46:12 +0200481
482 # Initialize the page counter properly based on nextStartIndex and any previously fetched results
483 # We add 1 to make it 1-based for display purposes since users expect page numbers to start from 1
484 # For first call, this will be 1, for subsequent calls, it will reflect our actual position
485 current_page_number <- ceiling(offset / maxResultsPerPage) + 1
486
487 # For sequential fetches, keep track of which global page we're on
488 # This is important for correctly showing page numbers in subsequent fetchNext calls
489 page_count_start <- current_page_number
490
Marc Kupietz5bbc9db2019-08-30 16:30:45 +0200491 repeat {
Marc Kupietz623d7122025-05-25 12:46:12 +0200492 # Determine which page to fetch next
493 if (randomizePageOrder) {
494 # In randomized mode, get the page from our randomized list using the page_index
495 # Make sure we don't exceed the array bounds
496 if (page_index > length(pages)) {
497 break # No more pages to fetch in randomized mode
498 }
499 current_offset_page <- pages[page_index]
500 # For display purposes in randomized mode, show which page out of the total we're fetching
501 display_page_number <- page_index
502 } else {
503 # In sequential mode, use the current_page_number to calculate the offset
504 current_offset_page <- (current_page_number - 1)
505 display_page_number <- current_page_number
506 }
507
508 # Calculate the actual offset in tokens
509 currentOffset <- current_offset_page * maxResultsPerPage
510
Marc Kupietzef0e9392025-06-18 12:21:49 +0200511 # Build the query with the appropriate count and offset using httr2
512 count_param <- min(if (!is.na(maxFetch)) maxFetch - results else maxResultsPerPage, maxResultsPerPage)
Marc Kupietzecc86702025-06-24 12:12:51 +0200513
Marc Kupietzef0e9392025-06-18 12:21:49 +0200514 # Parse existing URL to preserve all query parameters
515 parsed_url <- httr2::url_parse(kqo@requestUrl)
516 existing_query <- parsed_url$query
Marc Kupietzecc86702025-06-24 12:12:51 +0200517
Marc Kupietzef0e9392025-06-18 12:21:49 +0200518 # Add/update count and offset parameters
519 existing_query$count <- count_param
520 existing_query$offset <- currentOffset
Marc Kupietzecc86702025-06-24 12:12:51 +0200521
Marc Kupietzef0e9392025-06-18 12:21:49 +0200522 # Rebuild the URL with all parameters
523 query <- httr2::url_modify(kqo@requestUrl, query = existing_query)
Marc Kupietz68170952021-06-30 09:37:21 +0200524 res <- apiCall(kqo@korapConnection, query)
525 if (length(res$matches) == 0) {
526 break
527 }
528
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200529 if ("fields" %in% colnames(res$matches) && (is.na(use_korap_api) || as.numeric(use_korap_api) >= 1.0)) {
Marc Kupietz16ccf112025-01-26 13:25:27 +0100530 log_info(verbose, "Using fields API: ")
Marc Kupietz05a60792024-12-07 16:23:31 +0100531 currentMatches <- res$matches$fields %>%
532 purrr::map(~ mutate(.x, value = repair_data_strcuture(value))) %>%
533 tibble::enframe() %>%
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200534 tidyr::unnest(cols = value) %>%
535 tidyr::pivot_wider(names_from = key, id_cols = name, names_repair = "unique") %>%
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200536 dplyr::select(-name)
Marc Kupietzd8851222025-05-01 10:57:19 +0200537 if ("snippet" %in% colnames(res$matches)) {
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200538 currentMatches$snippet <- res$matches$snippet
539 }
Marc Kupietz3cd2c6c2025-01-08 20:35:39 +0100540 if ("tokens" %in% colnames(res$matches)) {
541 currentMatches$tokens <- res$matches$tokens
542 }
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200543 } else {
544 currentMatches <- res$matches
545 }
546
Marc Kupietze95108e2019-09-18 13:23:58 +0200547 for (field in kqo@fields) {
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200548 if (!field %in% colnames(currentMatches)) {
549 currentMatches[, field] <- NA
Marc Kupietz5bbc9db2019-08-30 16:30:45 +0200550 }
551 }
Marc Kupietzf4881122024-12-17 14:55:39 +0100552 currentMatches <- currentMatches %>%
553 select(kqo@fields) %>%
554 mutate(
Marc Kupietzff712a92025-07-18 09:07:23 +0200555 matchID = res$matches$matchID,
Marc Kupietz0447da02025-01-08 20:51:09 +0100556 tmp_positions = gsub(".*-p(\\d+)-(\\d+).*", "\\1 \\2", res$matches$matchID),
Marc Kupietzf4881122024-12-17 14:55:39 +0100557 matchStart = as.integer(stringr::word(tmp_positions, 1)),
558 matchEnd = as.integer(stringr::word(tmp_positions, 2)) - 1
559 ) %>%
560 select(-tmp_positions)
561
Marc Kupietz62da2b52019-09-12 17:43:34 +0200562 if (!is.list(collectedMatches)) {
563 collectedMatches <- currentMatches
Marc Kupietz5bbc9db2019-08-30 16:30:45 +0200564 } else {
Marc Kupietz2078bde2023-08-27 16:46:15 +0200565 collectedMatches <- bind_rows(collectedMatches, currentMatches)
Marc Kupietz5bbc9db2019-08-30 16:30:45 +0200566 }
Marc Kupietzae9b6172025-05-02 15:50:01 +0200567
Marc Kupietz623d7122025-05-25 12:46:12 +0200568 # Get the actual items per page from the API response
569 # We now consistently use maxResultsPerPage instead
Marc Kupietzacbaab02025-05-01 10:56:35 +0200570
Marc Kupietz623d7122025-05-25 12:46:12 +0200571 # Calculate total pages consistently using fixed maxResultsPerPage
572 # This ensures consistent page counting across the function
573 total_pages <- ceiling(kqo@totalResults / maxResultsPerPage)
574
Marc Kupietz24799fd2025-06-25 14:15:36 +0200575 # Calculate ETA using the centralized function from logging.R
576 current_page <- if (randomizePageOrder) page_index else display_page_number
577 total_pages_to_fetch <- if (!is.na(maxFetch)) {
578 # Account for offset - we can only fetch from the remaining results after offset
579 remaining_results_after_offset <- max(0, kqo@totalResults - offset)
580 min(ceiling(maxFetch / maxResultsPerPage), ceiling(remaining_results_after_offset / maxResultsPerPage))
581 } else {
582 total_pages
583 }
Marc Kupietz365660e2025-06-25 15:09:55 +0200584
Marc Kupietz24799fd2025-06-25 14:15:36 +0200585 eta_info <- calculate_eta(current_page, total_pages_to_fetch, start_time)
Marc Kupietz365660e2025-06-25 15:09:55 +0200586
Marc Kupietz24799fd2025-06-25 14:15:36 +0200587 # Extract timing information for display
Marc Kupietzae9b6172025-05-02 15:50:01 +0200588 time_per_page <- NA
Marc Kupietzae9b6172025-05-02 15:50:01 +0200589 if (!is.null(res$meta$benchmark) && is.character(res$meta$benchmark)) {
Marc Kupietzae9b6172025-05-02 15:50:01 +0200590 time_per_page <- suppressWarnings(as.numeric(sub("s", "", res$meta$benchmark)))
Marc Kupietzacbaab02025-05-01 10:56:35 +0200591 }
592
Marc Kupietz623d7122025-05-25 12:46:12 +0200593 # Create the page display string with proper formatting
Marc Kupietzacbaab02025-05-01 10:56:35 +0200594
Marc Kupietz623d7122025-05-25 12:46:12 +0200595 # For global page tracking, calculate the absolute page number
596 actual_display_number <- if (randomizePageOrder) {
597 current_offset_page + 1 # In randomized mode, this is the actual page (0-based + 1)
598 } else {
599 # In sequential mode, the absolute page number is the actual offset page + 1 (to make it 1-based)
600 current_offset_page + 1
601 }
602
603 # For subsequent calls to fetchNext, we need to calculate the correct page numbers
604 # based on the current batch being fetched
605
606 # For each call to fetchNext, we want to show 1/2, 2/2 (not 3/4, 4/4)
607 # Simply count from 1 within the current batch
608
609 # The relative page number is simply the current position in this batch
610 if (randomizePageOrder) {
611 relative_page_number <- page_index # In randomized mode, we start from 1 in each batch
612 } else {
613 relative_page_number <- display_page_number - (page_count_start - 1)
614 }
615
616 # How many pages will we fetch in this batch?
Marc Kupietz021663d2025-06-18 17:49:22 +0200617 # If maxFetch is specified, calculate the total pages for this fetch operation
Marc Kupietz623d7122025-05-25 12:46:12 +0200618 pages_in_this_batch <- if (!is.na(maxFetch)) {
Marc Kupietz021663d2025-06-18 17:49:22 +0200619 # Account for offset - we can only fetch from the remaining results after offset
620 remaining_results_after_offset <- max(0, kqo@totalResults - offset)
621 min(ceiling(maxFetch / maxResultsPerPage), ceiling(remaining_results_after_offset / maxResultsPerPage))
Marc Kupietz623d7122025-05-25 12:46:12 +0200622 } else {
623 # Otherwise fetch all remaining pages
624 total_pages - page_count_start + 1
625 }
626
627 # The total pages to be shown in this batch
628 batch_total_pages <- pages_in_this_batch
629
630 page_display <- paste0(
631 "Retrieved page ",
632 sprintf(paste0("%", nchar(batch_total_pages), "d"), relative_page_number),
633 "/",
634 sprintf("%d", batch_total_pages)
635 )
636
637 # If randomized, also show which actual page we fetched
638 if (randomizePageOrder) {
639 # Determine the maximum width needed for page numbers (based on total pages)
640 # This ensures consistent alignment
641 max_page_width <- nchar(as.character(total_pages))
642 # Add the actual page number that was fetched (0-based + 1 for display) with proper padding
Marc Kupietz7638ca42025-05-25 13:18:16 +0200643 page_display <- paste0(
644 page_display,
645 sprintf(" (actual page %*d)", max_page_width, current_offset_page + 1)
646 )
Marc Kupietz623d7122025-05-25 12:46:12 +0200647 }
648 # Always show the absolute page number and total pages (for clarity)
649 else {
650 # Show the absolute page number (out of total possible pages)
651 page_display <- paste0(page_display, sprintf(
652 " (page %d of %d total)",
653 actual_display_number, total_pages
654 ))
655 }
656
657 # Add caching or timing information
658 if (!is.null(res$meta$cached)) {
659 page_display <- paste0(page_display, " [cached]")
660 } else {
661 page_display <- paste0(
662 page_display,
663 " in ",
664 if (!is.na(time_per_page)) sprintf("%4.1f", time_per_page) else "?",
Marc Kupietz24799fd2025-06-25 14:15:36 +0200665 "s",
666 eta_info
Marc Kupietz623d7122025-05-25 12:46:12 +0200667 )
668 }
669
670 log_info(verbose, paste0(page_display, "\n"))
671
672 # Increment the appropriate counter based on mode
673 if (randomizePageOrder) {
674 page_index <- page_index + 1
675 } else {
676 current_page_number <- current_page_number + 1
677 }
Marc Kupietz5bbc9db2019-08-30 16:30:45 +0200678 results <- results + res$meta$itemsPerPage
Marc Kupietze8bd49b2024-06-28 07:24:44 +0200679 if (nrow(collectedMatches) >= kqo@totalResults || (!is.na(maxFetch) && results >= maxFetch)) {
Marc Kupietz5bbc9db2019-08-30 16:30:45 +0200680 break
681 }
682 }
Marc Kupietz68170952021-06-30 09:37:21 +0200683 nextStartIndex <- min(res$meta$startIndex + res$meta$itemsPerPage, kqo@totalResults)
Marc Kupietzd8851222025-05-01 10:57:19 +0200684 KorAPQuery(
685 nextStartIndex = nextStartIndex,
Marc Kupietzd0d3e9b2019-09-24 17:36:03 +0200686 korapConnection = kqo@korapConnection,
Marc Kupietze95108e2019-09-18 13:23:58 +0200687 fields = kqo@fields,
688 requestUrl = kqo@requestUrl,
689 request = kqo@request,
Marc Kupietz68170952021-06-30 09:37:21 +0200690 totalResults = kqo@totalResults,
Marc Kupietze95108e2019-09-18 13:23:58 +0200691 vc = kqo@vc,
692 webUIRequestUrl = kqo@webUIRequestUrl,
Marc Kupietz68170952021-06-30 09:37:21 +0200693 hasMoreMatches = (kqo@totalResults > nextStartIndex),
Marc Kupietze95108e2019-09-18 13:23:58 +0200694 apiResponse = res,
Marc Kupietzd8851222025-05-01 10:57:19 +0200695 collectedMatches = collectedMatches
696 )
Marc Kupietze95108e2019-09-18 13:23:58 +0200697})
Marc Kupietz62da2b52019-09-12 17:43:34 +0200698
699#' Fetch all results of a KorAP query.
Marc Kupietz62da2b52019-09-12 17:43:34 +0200700#'
Marc Kupietz67edcb52021-09-20 21:54:24 +0200701#' **`fetchAll`** fetches all results of a KorAP query.
Marc Kupietza6e4ee62021-03-05 09:00:15 +0100702#'
Marc Kupietza8c40f42025-06-24 15:49:52 +0200703#' @family corpus search functions
Marc Kupietzdc880ac2025-06-24 20:34:43 +0200704#' @param kqo object obtained from [corpusQuery()]
705#' @param verbose print progress information if true
706#' @param ... further arguments passed to [fetchNext()]
707#' @return The updated `kqo` object with all results in `@collectedMatches`
Marc Kupietza8c40f42025-06-24 15:49:52 +0200708#'
Marc Kupietz62da2b52019-09-12 17:43:34 +0200709#' @examples
Marc Kupietz6ae76052021-09-21 10:34:00 +0200710#' \dontrun{
Marc Kupietzecc86702025-06-24 12:12:51 +0200711#' # Fetch all metadata of every query hit for "Ameisenplage" and show a summary
712#' q <- KorAPConnection() |>
713#' corpusQuery("Ameisenplage") |>
Marc Kupietzd8851222025-05-01 10:57:19 +0200714#' fetchAll()
Marc Kupietze95108e2019-09-18 13:23:58 +0200715#' q@collectedMatches
Marc Kupietzecc86702025-06-24 12:12:51 +0200716#'
717#' # Fetch also all KWICs
718#' q <- KorAPConnection() |> auth() |>
719#' corpusQuery("Ameisenplage", metadataOnly = FALSE) |>
720#' fetchAll()
721#' q@collectedMatches
722#'
723#' # Retrieve title and text sigle metadata of all texts published on 1958-03-12
724#' q <- KorAPConnection() |>
725#' corpusQuery("<base/s=t>", # this matches each text once
726#' vc = "pubDate in 1958-03-12",
727#' fields = c("textSigle", "title"),
728#' ) |>
729#' fetchAll()
730#' q@collectedMatches
Marc Kupietz05b22772020-02-18 21:58:42 +0100731#' }
Marc Kupietz62da2b52019-09-12 17:43:34 +0200732#'
Marc Kupietze95108e2019-09-18 13:23:58 +0200733#' @aliases fetchAll
Marc Kupietz62da2b52019-09-12 17:43:34 +0200734#' @export
Marc Kupietzdbd431a2021-08-29 12:17:45 +0200735setMethod("fetchAll", "KorAPQuery", function(kqo, verbose = kqo@korapConnection@verbose, ...) {
736 return(fetchNext(kqo, offset = 0, maxFetch = NA, verbose = verbose, ...))
Marc Kupietze95108e2019-09-18 13:23:58 +0200737})
738
739#' Fetches the remaining results of a KorAP query.
740#'
Marc Kupietzdc880ac2025-06-24 20:34:43 +0200741#' @param kqo object obtained from [corpusQuery()]
742#' @param verbose print progress information if true
743#' @param ... further arguments passed to [fetchNext()]
744#' @return The updated `kqo` object with remaining results in `@collectedMatches`
745#'
Marc Kupietze95108e2019-09-18 13:23:58 +0200746#' @examples
Marc Kupietz6ae76052021-09-21 10:34:00 +0200747#' \dontrun{
748#'
Marc Kupietzd3526422025-06-25 09:16:15 +0200749#' q <- KorAPConnection() |>
750#' corpusQuery("Ameisenplage") |>
Marc Kupietzd8851222025-05-01 10:57:19 +0200751#' fetchRest()
Marc Kupietze95108e2019-09-18 13:23:58 +0200752#' q@collectedMatches
Marc Kupietz05b22772020-02-18 21:58:42 +0100753#' }
Marc Kupietze95108e2019-09-18 13:23:58 +0200754#'
755#' @aliases fetchRest
Marc Kupietze95108e2019-09-18 13:23:58 +0200756#' @export
Marc Kupietzdbd431a2021-08-29 12:17:45 +0200757setMethod("fetchRest", "KorAPQuery", function(kqo, verbose = kqo@korapConnection@verbose, ...) {
758 return(fetchNext(kqo, maxFetch = NA, verbose = verbose, ...))
Marc Kupietze95108e2019-09-18 13:23:58 +0200759})
760
Marc Kupietzbdedd022025-10-09 14:14:15 +0200761# Helper to collapse multiple annotation values while preserving order
762collapse_features <- function(values) {
763 if (length(values) == 0) {
764 return(NA_character_)
765 }
766 unique_values <- values[!duplicated(values)]
767 paste(unique_values, collapse = "|")
768}
769
770# Extract token-level annotations from a DOM node
771collect_token_annotations <- function(parent_node) {
772 if (inherits(parent_node, "xml_missing")) {
773 return(list(
774 node = list(),
775 token = character(0),
776 lemma = character(0),
777 pos = character(0),
778 morph = character(0)
779 ))
780 }
781
782 leaf_nodes <- xml2::xml_find_all(parent_node, ".//span[not(.//span)]")
783
784 if (length(leaf_nodes) == 0) {
785 return(list(
786 node = list(),
787 token = character(0),
788 lemma = character(0),
789 pos = character(0),
790 morph = character(0)
791 ))
792 }
793
794 tokens <- character(0)
795 lemmas <- character(0)
796 pos_tags <- character(0)
797 morph_tags <- character(0)
798 kept_nodes <- list()
799
800 for (idx in seq_along(leaf_nodes)) {
801 leaf <- leaf_nodes[[idx]]
802 token_text <- trimws(xml2::xml_text(leaf))
803 if (identical(token_text, "")) {
804 next
805 }
806
807 kept_nodes[[length(kept_nodes) + 1]] <- leaf
808 tokens <- c(tokens, token_text)
809
810 ancestors <- xml2::xml_find_all(leaf, "ancestor-or-self::span")
811 titles <- xml2::xml_attr(ancestors, "title")
812 titles <- titles[!is.na(titles)]
813
814 feature_pieces <- if (length(titles) > 0) unlist(strsplit(titles, "[[:space:]]+")) else character(0)
815
816 lemma_values <- sub('.*?/l:(.*)$', '\\1', feature_pieces[grepl('/l:', feature_pieces)], perl = TRUE)
817 pos_values <- sub('.*?/p:(.*)$', '\\1', feature_pieces[grepl('/p:', feature_pieces)], perl = TRUE)
818 morph_values <- sub('.*?/m:(.*)$', '\\1', feature_pieces[grepl('/m:', feature_pieces)], perl = TRUE)
819
820 lemmas <- c(lemmas, collapse_features(lemma_values))
821 pos_tags <- c(pos_tags, collapse_features(pos_values))
822 morph_tags <- c(morph_tags, collapse_features(morph_values))
823 }
824
825 list(
826 node = kept_nodes,
827 token = tokens,
828 lemma = lemmas,
829 pos = pos_tags,
830 morph = morph_tags
831 )
832}
833
Marc Kupietza29f3d42025-07-18 10:14:43 +0200834#'
835#' Parse XML annotations into linguistic layers
836#'
837#' Internal helper function to extract linguistic annotations (lemma, POS, morphology)
838#' from XML annotation snippets returned by the KorAP API.
839#'
840#' @param xml_snippet XML string containing annotation data
841#' @return Named list with vectors for 'token', 'lemma', 'pos', and 'morph'
842#' @keywords internal
843parse_xml_annotations <- function(xml_snippet) {
844 if (is.null(xml_snippet) || is.na(xml_snippet) || xml_snippet == "") {
845 return(list(token = character(0), lemma = character(0), pos = character(0), morph = character(0)))
846 }
847
Marc Kupietzbdedd022025-10-09 14:14:15 +0200848 doc <- tryCatch(xml2::read_html(paste0("<root>", xml_snippet, "</root>")), error = function(e) NULL)
849 if (is.null(doc)) {
850 return(list(token = character(0), lemma = character(0), pos = character(0), morph = character(0)))
Marc Kupietzcd452182025-10-09 13:28:41 +0200851 }
852
Marc Kupietzbdedd022025-10-09 14:14:15 +0200853 match_node <- xml2::xml_find_first(doc, ".//span[contains(@class, 'match')]")
854 if (inherits(match_node, "xml_missing")) {
855 match_node <- xml2::xml_find_first(doc, ".//span")
856 if (inherits(match_node, "xml_missing")) {
857 return(list(token = character(0), lemma = character(0), pos = character(0), morph = character(0)))
Marc Kupietza29f3d42025-07-18 10:14:43 +0200858 }
859 }
860
Marc Kupietzbdedd022025-10-09 14:14:15 +0200861 token_info <- collect_token_annotations(match_node)
Marc Kupietza29f3d42025-07-18 10:14:43 +0200862
Marc Kupietzbdedd022025-10-09 14:14:15 +0200863 list(
864 token = token_info$token,
865 lemma = token_info$lemma,
866 pos = token_info$pos,
867 morph = token_info$morph
868 )
Marc Kupietza29f3d42025-07-18 10:14:43 +0200869}
870
871#'
872#' Parse XML annotations into linguistic layers with left/match/right structure
873#'
874#' Internal helper function to extract linguistic annotations (lemma, POS, morphology)
875#' from XML annotation snippets returned by the KorAP API, split into left context,
876#' match, and right context sections like the tokens field.
877#'
878#' @param xml_snippet XML string containing annotation data
879#' @return Named list with nested structure containing left/match/right for 'atokens', 'lemma', 'pos', and 'morph'
880#' @keywords internal
881parse_xml_annotations_structured <- function(xml_snippet) {
882 if (is.null(xml_snippet) || is.na(xml_snippet) || xml_snippet == "") {
883 empty_result <- list(left = character(0), match = character(0), right = character(0))
884 return(list(
885 atokens = empty_result,
886 lemma = empty_result,
887 pos = empty_result,
888 morph = empty_result
889 ))
890 }
891
Marc Kupietzbdedd022025-10-09 14:14:15 +0200892 doc <- tryCatch(xml2::read_html(paste0("<root>", xml_snippet, "</root>")), error = function(e) NULL)
893 if (is.null(doc)) {
894 empty_result <- list(left = character(0), match = character(0), right = character(0))
Marc Kupietza29f3d42025-07-18 10:14:43 +0200895 return(list(
Marc Kupietzbdedd022025-10-09 14:14:15 +0200896 atokens = empty_result,
897 lemma = empty_result,
898 pos = empty_result,
899 morph = empty_result
Marc Kupietza29f3d42025-07-18 10:14:43 +0200900 ))
901 }
902
Marc Kupietzbdedd022025-10-09 14:14:15 +0200903 match_node <- xml2::xml_find_first(doc, ".//span[contains(@class, 'match')]")
904 if (inherits(match_node, "xml_missing")) {
905 empty_result <- list(left = character(0), match = character(0), right = character(0))
906 return(list(
907 atokens = empty_result,
908 lemma = empty_result,
909 pos = empty_result,
910 morph = empty_result
911 ))
Marc Kupietza29f3d42025-07-18 10:14:43 +0200912 }
Marc Kupietzc643a122025-07-18 18:18:36 +0200913
Marc Kupietzbdedd022025-10-09 14:14:15 +0200914 token_info <- collect_token_annotations(match_node)
915 tokens <- token_info$token
916 lemmas <- token_info$lemma
917 pos_tags <- token_info$pos
918 morph_tags <- token_info$morph
919 nodes <- token_info$node
Marc Kupietzc643a122025-07-18 18:18:36 +0200920
Marc Kupietzbdedd022025-10-09 14:14:15 +0200921 if (length(tokens) == 0) {
922 empty_result <- list(left = character(0), match = character(0), right = character(0))
923 return(list(
924 atokens = empty_result,
925 lemma = empty_result,
926 pos = empty_result,
927 morph = empty_result
928 ))
929 }
Marc Kupietzc643a122025-07-18 18:18:36 +0200930
Marc Kupietzbdedd022025-10-09 14:14:15 +0200931 mark_flags <- vapply(nodes, function(n) {
932 !inherits(xml2::xml_find_first(n, "ancestor::mark"), "xml_missing")
933 }, logical(1))
Marc Kupietzc643a122025-07-18 18:18:36 +0200934
Marc Kupietzbdedd022025-10-09 14:14:15 +0200935 if (any(mark_flags)) {
936 first_idx <- which(mark_flags)[1]
937 last_idx <- tail(which(mark_flags), 1)
Marc Kupietza29f3d42025-07-18 10:14:43 +0200938 } else {
Marc Kupietzbdedd022025-10-09 14:14:15 +0200939 first_idx <- 1
940 last_idx <- length(tokens)
Marc Kupietza29f3d42025-07-18 10:14:43 +0200941 }
942
Marc Kupietzbdedd022025-10-09 14:14:15 +0200943 sections <- rep("match", length(tokens))
944 if (first_idx > 1) {
945 sections[seq_len(first_idx - 1)] <- "left"
946 }
947 if (last_idx < length(tokens)) {
948 sections[seq(from = last_idx + 1, to = length(tokens))] <- "right"
949 }
Marc Kupietza29f3d42025-07-18 10:14:43 +0200950
Marc Kupietzbdedd022025-10-09 14:14:15 +0200951 subset_by_section <- function(values, section) {
952 idx <- sections == section
953 if (!any(idx)) {
954 return(character(0))
955 }
956 values[idx]
957 }
958
959 atokens <- list(
960 left = subset_by_section(tokens, "left"),
961 match = subset_by_section(tokens, "match"),
962 right = subset_by_section(tokens, "right")
963 )
964
965 lemma <- list(
966 left = subset_by_section(lemmas, "left"),
967 match = subset_by_section(lemmas, "match"),
968 right = subset_by_section(lemmas, "right")
969 )
970
971 pos <- list(
972 left = subset_by_section(pos_tags, "left"),
973 match = subset_by_section(pos_tags, "match"),
974 right = subset_by_section(pos_tags, "right")
975 )
976
977 morph <- list(
978 left = subset_by_section(morph_tags, "left"),
979 match = subset_by_section(morph_tags, "match"),
980 right = subset_by_section(morph_tags, "right")
981 )
982
983 list(
984 atokens = atokens,
985 lemma = lemma,
986 pos = pos,
987 morph = morph
988 )
Marc Kupietza29f3d42025-07-18 10:14:43 +0200989}
990
Marc Kupietze52b2952025-07-17 16:53:02 +0200991#' Fetch annotations for all collected matches
992#'
Marc Kupietz89f796e2025-07-19 09:05:06 +0200993#' `r lifecycle::badge("experimental")`
994#'
995#' **`fetchAnnotations`** fetches annotations (only token annotations, for now)
996#' for all matches in the `@collectedMatches` slot
Marc Kupietzc643a122025-07-18 18:18:36 +0200997#' of a KorAPQuery object and adds annotation columns directly to the `@collectedMatches`
Marc Kupietz89f796e2025-07-19 09:05:06 +0200998#' data frame. The method uses the `matchID` from collected matches.
Marc Kupietza29f3d42025-07-18 10:14:43 +0200999#'
1000#' **Important**: For copyright-restricted corpora, users must be authorized via [auth()]
1001#' and the initial corpus query must have `metadataOnly = FALSE` to ensure snippets are
1002#' available for annotation parsing.
1003#'
1004#' The method parses XML snippet annotations and adds linguistic columns to the data frame:
1005#' - `pos`: data frame with `left`, `match`, `right` columns, each containing list vectors of part-of-speech tags
1006#' - `lemma`: data frame with `left`, `match`, `right` columns, each containing list vectors of lemmas
1007#' - `morph`: data frame with `left`, `match`, `right` columns, each containing list vectors of morphological tags
1008#' - `atokens`: data frame with `left`, `match`, `right` columns, each containing list vectors of token text (from annotations)
1009#' - `annotation_snippet`: original XML snippet from the annotation API
Marc Kupietze52b2952025-07-17 16:53:02 +02001010#'
1011#' @family corpus search functions
Marc Kupietz89f796e2025-07-19 09:05:06 +02001012#' @concept Annotations
Marc Kupietze52b2952025-07-17 16:53:02 +02001013#'
Marc Kupietza29f3d42025-07-18 10:14:43 +02001014#' @param kqo object obtained from [corpusQuery()] with collected matches. Note: the original corpus query should have `metadataOnly = FALSE` for annotation parsing to work.
Marc Kupietze52b2952025-07-17 16:53:02 +02001015#' @param foundry string specifying the foundry to use for annotations (default: "tt" for Tree-Tagger)
Marc Kupietz93787d52025-09-03 13:33:25 +02001016#' @param overwrite logical; if TRUE, re-fetch and replace any existing
1017#' annotation columns. If FALSE (default), only add missing annotation layers
1018#' and preserve already fetched ones (e.g., keep POS/lemma from a previous
1019#' foundry while adding morph from another).
Marc Kupietze52b2952025-07-17 16:53:02 +02001020#' @param verbose print progress information if true
Marc Kupietz0af75932025-09-09 18:14:16 +02001021#' @return The updated `kqo` object with annotation columns
Marc Kupietz89f796e2025-07-19 09:05:06 +02001022#' like `pos`, `lemma`, `morph` (and `atokens` and `annotation_snippet`)
1023#' in the `@collectedMatches` slot. Each column is a data frame
1024#' with `left`, `match`, and `right` columns containing list vectors of annotations
1025#' for the left context, matched tokens, and right context, respectively.
1026#' The original XML snippet for each match is also stored in `annotation_snippet`.
Marc Kupietze52b2952025-07-17 16:53:02 +02001027#'
1028#' @examples
1029#' \dontrun{
1030#'
1031#' # Fetch annotations for matches using Tree-Tagger foundry
Marc Kupietza29f3d42025-07-18 10:14:43 +02001032#' # Note: Authorization required for copyright-restricted corpora
Marc Kupietze52b2952025-07-17 16:53:02 +02001033#' q <- KorAPConnection() |>
Marc Kupietza29f3d42025-07-18 10:14:43 +02001034#' auth() |>
1035#' corpusQuery("Ameisenplage", metadataOnly = FALSE) |>
Marc Kupietze52b2952025-07-17 16:53:02 +02001036#' fetchNext(maxFetch = 10) |>
1037#' fetchAnnotations()
Marc Kupietze52b2952025-07-17 16:53:02 +02001038#'
Marc Kupietza29f3d42025-07-18 10:14:43 +02001039#' # Access linguistic annotations for match i:
Marc Kupietz6aa5a0d2025-09-08 17:51:47 +02001040#' pos_tags <- q@collectedMatches$pos
1041#' # Data frame with left/match/right columns for POS tags
1042#' lemmas <- q@collectedMatches$lemma
1043#' # Data frame with left/match/right columns for lemmas
1044#' morphology <- q@collectedMatches$morph
1045#' # Data frame with left/match/right columns for morphological tags
1046#' atokens <- q@collectedMatches$atokens
1047#' # Data frame with left/match/right columns for annotation token text
Marc Kupietz0af75932025-09-09 18:14:16 +02001048#' # Original XML snippet for match i
1049#' raw_snippet <- q@collectedMatches$annotation_snippet[[i]]
Marc Kupietzc643a122025-07-18 18:18:36 +02001050#'
Marc Kupietza29f3d42025-07-18 10:14:43 +02001051#' # Access specific components:
Marc Kupietz0af75932025-09-09 18:14:16 +02001052#' # POS tags for the matched tokens in match i
1053#' match_pos <- q@collectedMatches$pos$match[[i]]
1054#' # Lemmas for the left context in match i
1055#' left_lemmas <- q@collectedMatches$lemma$left[[i]]
1056#' # Token text for the right context in match i
1057#' right_tokens <- q@collectedMatches$atokens$right[[i]]
Marc Kupietza29f3d42025-07-18 10:14:43 +02001058#'
Marc Kupietz89f796e2025-07-19 09:05:06 +02001059#' # Use a different foundry (e.g., MarMoT)
Marc Kupietze52b2952025-07-17 16:53:02 +02001060#' q <- KorAPConnection() |>
Marc Kupietza29f3d42025-07-18 10:14:43 +02001061#' auth() |>
1062#' corpusQuery("Ameisenplage", metadataOnly = FALSE) |>
Marc Kupietze52b2952025-07-17 16:53:02 +02001063#' fetchNext(maxFetch = 10) |>
Marc Kupietz89f796e2025-07-19 09:05:06 +02001064#' fetchAnnotations(foundry = "marmot")
1065#' q@collectedMatches$pos$left[1] # POS tags for the left context of the first match
Marc Kupietze52b2952025-07-17 16:53:02 +02001066#' }
Marc Kupietze52b2952025-07-17 16:53:02 +02001067#' @export
Marc Kupietz0af75932025-09-09 18:14:16 +02001068setMethod("fetchAnnotations", "KorAPQuery", function(kqo,
1069 foundry = "tt",
1070 overwrite = FALSE,
1071 verbose = kqo@korapConnection@verbose) {
1072 if (is.null(kqo@collectedMatches) ||
1073 nrow(kqo@collectedMatches) == 0) {
1074 warning("No collected matches found. Please run fetchNext() or fetchAll() first.")
1075 return(kqo)
1076 }
Marc Kupietza29f3d42025-07-18 10:14:43 +02001077
Marc Kupietze52b2952025-07-17 16:53:02 +02001078 df <- kqo@collectedMatches
1079 kco <- kqo@korapConnection
Marc Kupietza29f3d42025-07-18 10:14:43 +02001080
Marc Kupietza29f3d42025-07-18 10:14:43 +02001081 # Initialize annotation columns as data frames (like tokens field)
1082 # Create the structure more explicitly to avoid assignment issues
1083 nrows <- nrow(df)
Marc Kupietzc643a122025-07-18 18:18:36 +02001084
Marc Kupietz03d2b1a2025-07-19 09:14:45 +02001085 # Pre-compute the empty character vector list to avoid repeated computation
1086 empty_char_list <- I(replicate(nrows, character(0), simplify = FALSE))
Marc Kupietz0af75932025-09-09 18:14:16 +02001087
Marc Kupietz03d2b1a2025-07-19 09:14:45 +02001088 # Helper function to create annotation data frame structure
1089 create_annotation_df <- function(empty_list) {
1090 data.frame(
1091 left = empty_list,
1092 match = empty_list,
1093 right = empty_list,
1094 stringsAsFactors = FALSE
1095 )
1096 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001097
Marc Kupietz93787d52025-09-03 13:33:25 +02001098 # Track which annotation columns already existed to decide overwrite behavior
1099 existing_types <- list(
1100 pos = "pos" %in% colnames(df),
1101 lemma = "lemma" %in% colnames(df),
1102 morph = "morph" %in% colnames(df),
1103 atokens = "atokens" %in% colnames(df),
1104 annotation_snippet = "annotation_snippet" %in% colnames(df)
1105 )
1106
1107 # Initialize annotation columns using the helper function
Marc Kupietz03d2b1a2025-07-19 09:14:45 +02001108 annotation_types <- c("pos", "lemma", "morph", "atokens")
1109 for (type in annotation_types) {
Marc Kupietz93787d52025-09-03 13:33:25 +02001110 if (overwrite || !existing_types[[type]]) {
1111 df[[type]] <- create_annotation_df(empty_char_list)
1112 }
Marc Kupietz03d2b1a2025-07-19 09:14:45 +02001113 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001114
Marc Kupietz93787d52025-09-03 13:33:25 +02001115 if (overwrite || !existing_types$annotation_snippet) {
feldmuellera02f1932025-09-15 16:38:06 +02001116 df$annotation_snippet <- rep(NA_character_, nrows) # Fixed line
Marc Kupietz93787d52025-09-03 13:33:25 +02001117 }
Marc Kupietza29f3d42025-07-18 10:14:43 +02001118
Marc Kupietze8c0fef2025-07-18 19:59:04 +02001119 # Initialize timing for ETA calculation
1120 start_time <- Sys.time()
1121 if (verbose) {
1122 log_info(verbose, paste("Starting to fetch annotations for", nrows, "matches\n"))
1123 }
1124
Marc Kupietz93787d52025-09-03 13:33:25 +02001125 # Helper to decide if existing annotation row is effectively empty
1126 is_empty_annotation_row <- function(ann_df, row_index) {
1127 if (is.null(ann_df) || nrow(ann_df) < row_index) return(TRUE)
1128 left_val <- ann_df$left[[row_index]]
1129 match_val <- ann_df$match[[row_index]]
1130 right_val <- ann_df$right[[row_index]]
1131 all(
1132 (is.null(left_val) || (length(left_val) == 0) || all(is.na(left_val))),
1133 (is.null(match_val) || (length(match_val) == 0) || all(is.na(match_val))),
1134 (is.null(right_val) || (length(right_val) == 0) || all(is.na(right_val)))
1135 )
1136 }
1137
Marc Kupietze52b2952025-07-17 16:53:02 +02001138 for (i in seq_len(nrow(df))) {
Marc Kupietze8c0fef2025-07-18 19:59:04 +02001139 # ETA logging
1140 if (verbose && i > 1) {
1141 eta_info <- calculate_eta(i, nrows, start_time)
1142 log_info(verbose, paste("Fetching annotations for match", i, "of", nrows, eta_info, "\n"))
1143 }
Marc Kupietzff712a92025-07-18 09:07:23 +02001144 # Use matchID if available, otherwise fall back to constructing from matchStart/matchEnd
1145 if ("matchID" %in% colnames(df) && !is.na(df$matchID[i])) {
Marc Kupietza29f3d42025-07-18 10:14:43 +02001146 # matchID format: "match-match-A00/JUN/39609-p202-203" or encrypted format like
1147 # "match-DNB10/CSL/80400-p2343-2344x_MinDOhu_P6dd2MMZJyyus_7MairdKnr1LxY07Cya-Ow"
1148 # Extract document path and position, handling both regular and encrypted formats
Marc Kupietzc643a122025-07-18 18:18:36 +02001149
Marc Kupietza29f3d42025-07-18 10:14:43 +02001150 # More flexible regex to extract the document path with position and encryption
1151 # Look for pattern: match-(...)-p(\d+)-(\d+)(.*) where (.*) is the encrypted part
1152 # We need to capture the entire path including the encrypted suffix
1153 match_result <- regexpr("match-(.+?-p\\d+-\\d+.*)", df$matchID[i], perl = TRUE)
Marc Kupietzc643a122025-07-18 18:18:36 +02001154
Marc Kupietza29f3d42025-07-18 10:14:43 +02001155 if (match_result > 0) {
1156 # Extract the complete path including encryption (everything after "match-")
1157 doc_path_with_pos_and_encryption <- gsub("^match-(.+)$", "\\1", df$matchID[i], perl = TRUE)
1158 # Convert the dash before position to slash, but keep everything after the position
1159 match_path <- gsub("-p(\\d+-\\d+.*)", "/p\\1", doc_path_with_pos_and_encryption)
Marc Kupietz25121302025-07-19 08:45:43 +02001160 # Use httr2 to construct URL safely
1161 base_url <- paste0(kco@apiUrl, "corpus/", match_path)
1162 req <- httr2::url_modify(base_url, query = list(foundry = foundry))
Marc Kupietza29f3d42025-07-18 10:14:43 +02001163 } else {
Marc Kupietz25121302025-07-19 08:45:43 +02001164 # If regex fails, fall back to the old method with httr2
1165 # Format numbers to avoid scientific notation
1166 match_start <- format(df$matchStart[i], scientific = FALSE)
1167 match_end <- format(df$matchEnd[i], scientific = FALSE)
1168 base_url <- paste0(kco@apiUrl, "corpus/", df$textSigle[i], "/", "p", match_start, "-", match_end)
1169 req <- httr2::url_modify(base_url, query = list(foundry = foundry))
Marc Kupietzff712a92025-07-18 09:07:23 +02001170 }
1171 } else {
Marc Kupietz25121302025-07-19 08:45:43 +02001172 # Fallback to the old method with httr2
1173 # Format numbers to avoid scientific notation
1174 match_start <- format(df$matchStart[i], scientific = FALSE)
1175 match_end <- format(df$matchEnd[i], scientific = FALSE)
1176 base_url <- paste0(kco@apiUrl, "corpus/", df$textSigle[i], "/", "p", match_start, "-", match_end)
1177 req <- httr2::url_modify(base_url, query = list(foundry = foundry))
Marc Kupietzff712a92025-07-18 09:07:23 +02001178 }
Marc Kupietza29f3d42025-07-18 10:14:43 +02001179
Marc Kupietze52b2952025-07-17 16:53:02 +02001180 tryCatch({
1181 res <- apiCall(kco, req)
Marc Kupietzc643a122025-07-18 18:18:36 +02001182
Marc Kupietze52b2952025-07-17 16:53:02 +02001183 if (!is.null(res)) {
Marc Kupietz93787d52025-09-03 13:33:25 +02001184 # Store the raw annotation snippet (respect overwrite flag)
1185 if (overwrite || !existing_types$annotation_snippet || is.null(df$annotation_snippet[[i]]) || is.na(df$annotation_snippet[[i]])) {
1186 df$annotation_snippet[[i]] <- if (is.list(res) && "snippet" %in% names(res)) res$snippet else NA
1187 }
Marc Kupietza29f3d42025-07-18 10:14:43 +02001188
1189 # Parse XML annotations if snippet is available
1190 if (is.list(res) && "snippet" %in% names(res)) {
1191 parsed_annotations <- parse_xml_annotations_structured(res$snippet)
1192
1193 # Store the parsed linguistic data in data frame format (like tokens)
1194 # Use individual assignment to avoid data frame mismatch errors
1195 tryCatch({
1196 # Assign POS annotations
Marc Kupietz93787d52025-09-03 13:33:25 +02001197 if (overwrite || !existing_types$pos || is_empty_annotation_row(df$pos, i)) {
1198 df$pos$left[i] <- list(parsed_annotations$pos$left)
1199 df$pos$match[i] <- list(parsed_annotations$pos$match)
1200 df$pos$right[i] <- list(parsed_annotations$pos$right)
1201 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001202
Marc Kupietza29f3d42025-07-18 10:14:43 +02001203 # Assign lemma annotations
Marc Kupietz93787d52025-09-03 13:33:25 +02001204 if (overwrite || !existing_types$lemma || is_empty_annotation_row(df$lemma, i)) {
1205 df$lemma$left[i] <- list(parsed_annotations$lemma$left)
1206 df$lemma$match[i] <- list(parsed_annotations$lemma$match)
1207 df$lemma$right[i] <- list(parsed_annotations$lemma$right)
1208 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001209
Marc Kupietza29f3d42025-07-18 10:14:43 +02001210 # Assign morphology annotations
Marc Kupietz93787d52025-09-03 13:33:25 +02001211 if (overwrite || !existing_types$morph || is_empty_annotation_row(df$morph, i)) {
1212 df$morph$left[i] <- list(parsed_annotations$morph$left)
1213 df$morph$match[i] <- list(parsed_annotations$morph$match)
1214 df$morph$right[i] <- list(parsed_annotations$morph$right)
1215 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001216
Marc Kupietza29f3d42025-07-18 10:14:43 +02001217 # Assign token annotations
Marc Kupietz93787d52025-09-03 13:33:25 +02001218 if (overwrite || !existing_types$atokens || is_empty_annotation_row(df$atokens, i)) {
1219 df$atokens$left[i] <- list(parsed_annotations$atokens$left)
1220 df$atokens$match[i] <- list(parsed_annotations$atokens$match)
1221 df$atokens$right[i] <- list(parsed_annotations$atokens$right)
1222 }
Marc Kupietza29f3d42025-07-18 10:14:43 +02001223 }, error = function(assign_error) {
Marc Kupietza29f3d42025-07-18 10:14:43 +02001224 # Set empty character vectors on assignment error using list assignment
Marc Kupietz93787d52025-09-03 13:33:25 +02001225 if (overwrite || !existing_types$pos) {
1226 df$pos$left[i] <<- list(character(0))
1227 df$pos$match[i] <<- list(character(0))
1228 df$pos$right[i] <<- list(character(0))
1229 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001230
Marc Kupietz93787d52025-09-03 13:33:25 +02001231 if (overwrite || !existing_types$lemma) {
1232 df$lemma$left[i] <<- list(character(0))
1233 df$lemma$match[i] <<- list(character(0))
1234 df$lemma$right[i] <<- list(character(0))
1235 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001236
Marc Kupietz93787d52025-09-03 13:33:25 +02001237 if (overwrite || !existing_types$morph) {
1238 df$morph$left[i] <<- list(character(0))
1239 df$morph$match[i] <<- list(character(0))
1240 df$morph$right[i] <<- list(character(0))
1241 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001242
Marc Kupietz93787d52025-09-03 13:33:25 +02001243 if (overwrite || !existing_types$atokens) {
1244 df$atokens$left[i] <<- list(character(0))
1245 df$atokens$match[i] <<- list(character(0))
1246 df$atokens$right[i] <<- list(character(0))
1247 }
Marc Kupietza29f3d42025-07-18 10:14:43 +02001248 })
Marc Kupietza29f3d42025-07-18 10:14:43 +02001249 } else {
1250 # No snippet available, store empty vectors
Marc Kupietz93787d52025-09-03 13:33:25 +02001251 if (overwrite || !existing_types$pos) {
1252 df$pos$left[i] <- list(character(0))
1253 df$pos$match[i] <- list(character(0))
1254 df$pos$right[i] <- list(character(0))
1255 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001256
Marc Kupietz93787d52025-09-03 13:33:25 +02001257 if (overwrite || !existing_types$lemma) {
1258 df$lemma$left[i] <- list(character(0))
1259 df$lemma$match[i] <- list(character(0))
1260 df$lemma$right[i] <- list(character(0))
1261 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001262
Marc Kupietz93787d52025-09-03 13:33:25 +02001263 if (overwrite || !existing_types$morph) {
1264 df$morph$left[i] <- list(character(0))
1265 df$morph$match[i] <- list(character(0))
1266 df$morph$right[i] <- list(character(0))
1267 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001268
Marc Kupietz93787d52025-09-03 13:33:25 +02001269 if (overwrite || !existing_types$atokens) {
1270 df$atokens$left[i] <- list(character(0))
1271 df$atokens$match[i] <- list(character(0))
1272 df$atokens$right[i] <- list(character(0))
1273 }
Marc Kupietza29f3d42025-07-18 10:14:43 +02001274 }
Marc Kupietze52b2952025-07-17 16:53:02 +02001275 } else {
Marc Kupietza29f3d42025-07-18 10:14:43 +02001276 # Store NAs for failed requests
Marc Kupietz93787d52025-09-03 13:33:25 +02001277 if (overwrite || !existing_types$pos) {
1278 df$pos$left[i] <- list(NA)
1279 df$pos$match[i] <- list(NA)
1280 df$pos$right[i] <- list(NA)
1281 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001282
Marc Kupietz93787d52025-09-03 13:33:25 +02001283 if (overwrite || !existing_types$lemma) {
1284 df$lemma$left[i] <- list(NA)
1285 df$lemma$match[i] <- list(NA)
1286 df$lemma$right[i] <- list(NA)
1287 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001288
Marc Kupietz93787d52025-09-03 13:33:25 +02001289 if (overwrite || !existing_types$morph) {
1290 df$morph$left[i] <- list(NA)
1291 df$morph$match[i] <- list(NA)
1292 df$morph$right[i] <- list(NA)
1293 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001294
Marc Kupietz93787d52025-09-03 13:33:25 +02001295 if (overwrite || !existing_types$atokens) {
1296 df$atokens$left[i] <- list(NA)
1297 df$atokens$match[i] <- list(NA)
1298 df$atokens$right[i] <- list(NA)
1299 }
1300 if (overwrite || !existing_types$annotation_snippet) {
1301 df$annotation_snippet[[i]] <- NA
1302 }
Marc Kupietze52b2952025-07-17 16:53:02 +02001303 }
1304 }, error = function(e) {
Marc Kupietza29f3d42025-07-18 10:14:43 +02001305 # Store NAs for failed requests
Marc Kupietz93787d52025-09-03 13:33:25 +02001306 if (overwrite || !existing_types$pos) {
1307 df$pos$left[i] <- list(NA)
1308 df$pos$match[i] <- list(NA)
1309 df$pos$right[i] <- list(NA)
1310 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001311
Marc Kupietz93787d52025-09-03 13:33:25 +02001312 if (overwrite || !existing_types$lemma) {
1313 df$lemma$left[i] <- list(NA)
1314 df$lemma$match[i] <- list(NA)
1315 df$lemma$right[i] <- list(NA)
1316 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001317
Marc Kupietz93787d52025-09-03 13:33:25 +02001318 if (overwrite || !existing_types$morph) {
1319 df$morph$left[i] <- list(NA)
1320 df$morph$match[i] <- list(NA)
1321 df$morph$right[i] <- list(NA)
1322 }
Marc Kupietzc643a122025-07-18 18:18:36 +02001323
Marc Kupietz93787d52025-09-03 13:33:25 +02001324 if (overwrite || !existing_types$atokens) {
1325 df$atokens$left[i] <- list(NA)
1326 df$atokens$match[i] <- list(NA)
1327 df$atokens$right[i] <- list(NA)
1328 }
1329 if (overwrite || !existing_types$annotation_snippet) {
1330 df$annotation_snippet[[i]] <- NA
1331 }
Marc Kupietze52b2952025-07-17 16:53:02 +02001332 })
1333 }
Marc Kupietza29f3d42025-07-18 10:14:43 +02001334
Marc Kupietza29f3d42025-07-18 10:14:43 +02001335 # Validate data frame structure before assignment
1336 if (nrow(df) != nrow(kqo@collectedMatches)) {
Marc Kupietza29f3d42025-07-18 10:14:43 +02001337 }
1338
1339 # Update the collectedMatches with annotation data
1340 tryCatch({
1341 kqo@collectedMatches <- df
1342 }, error = function(assign_error) {
Marc Kupietza29f3d42025-07-18 10:14:43 +02001343 # Try a safer approach: add columns individually
1344 tryCatch({
1345 kqo@collectedMatches$pos <- df$pos
Marc Kupietzc643a122025-07-18 18:18:36 +02001346 kqo@collectedMatches$lemma <- df$lemma
Marc Kupietza29f3d42025-07-18 10:14:43 +02001347 kqo@collectedMatches$morph <- df$morph
1348 kqo@collectedMatches$atokens <- df$atokens
1349 kqo@collectedMatches$annotation_snippet <- df$annotation_snippet
1350 }, error = function(col_error) {
Marc Kupietza29f3d42025-07-18 10:14:43 +02001351 warning("Failed to add annotation data to collectedMatches")
1352 })
1353 })
1354
Marc Kupietze8c0fef2025-07-18 19:59:04 +02001355 if (verbose) {
1356 elapsed_time <- Sys.time() - start_time
1357 log_info(verbose, paste("Finished fetching annotations for", nrows, "matches in", format_duration(as.numeric(elapsed_time, units = "secs")), "\n"))
1358 }
1359
Marc Kupietze52b2952025-07-17 16:53:02 +02001360 return(kqo)
1361})
1362
Marc Kupietzad8d2ed2025-04-05 15:37:38 +02001363#' Query frequencies of search expressions in virtual corpora
Marc Kupietz3f575282019-10-04 14:46:04 +02001364#'
Marc Kupietz67edcb52021-09-20 21:54:24 +02001365#' **`frequencyQuery`** combines [corpusQuery()], [corpusStats()] and
Marc Kupietzad8d2ed2025-04-05 15:37:38 +02001366#' [ci()] to compute a tibble with the absolute and relative frequencies and
Marc Kupietz3f575282019-10-04 14:46:04 +02001367#' confidence intervals of one ore multiple search terms across one or multiple
1368#' virtual corpora.
1369#'
Marc Kupietza8c40f42025-06-24 15:49:52 +02001370#' @family frequency analysis
Marc Kupietz3f575282019-10-04 14:46:04 +02001371#' @aliases frequencyQuery
Marc Kupietz3f575282019-10-04 14:46:04 +02001372#' @examples
Marc Kupietz6ae76052021-09-21 10:34:00 +02001373#' \dontrun{
1374#'
Marc Kupietzad8d2ed2025-04-05 15:37:38 +02001375#' KorAPConnection(verbose = TRUE) |>
Marc Kupietz3f575282019-10-04 14:46:04 +02001376#' frequencyQuery(c("Mücke", "Schnake"), paste0("pubDate in ", 2000:2003))
Marc Kupietz05b22772020-02-18 21:58:42 +01001377#' }
Marc Kupietz3f575282019-10-04 14:46:04 +02001378#'
Marc Kupietzad8d2ed2025-04-05 15:37:38 +02001379# @inheritParams corpusQuery
Marc Kupietz617266d2025-02-27 10:43:07 +01001380#' @param kco [KorAPConnection()] object (obtained e.g. from `KorAPConnection()`
Marc Kupietzad8d2ed2025-04-05 15:37:38 +02001381#' @param query corpus query string(s.) (can be a vector). The query language depends on the `ql` parameter. Either `query` must be provided or `KorAPUrl`.
1382#' @param vc virtual corpus definition(s) (can be a vector)
Marc Kupietz67edcb52021-09-20 21:54:24 +02001383#' @param conf.level confidence level of the returned confidence interval (passed through [ci()] to [prop.test()]).
1384#' @param as.alternatives LOGICAL that specifies if the query terms should be treated as alternatives. If `as.alternatives` is TRUE, the sum over all query hits, instead of the respective vc token sizes is used as total for the calculation of relative frequencies.
Marc Kupietza3a8cd92026-09-08 07:59:04 +02001385#' @param cacheAs path to an RDS file to keep the result in. If the file exists and records the same call, it is read back instead of contacting the server; otherwise the query is run and its result stored there. Unlike the connection's `cache`, this file belongs to the caller, which is what keeps an analysis reproducible once the corpus has grown or the scores have changed. Defaults to \code{NULL} (no file).
Marc Kupietzad8d2ed2025-04-05 15:37:38 +02001386#' @param ... further arguments passed to or from other methods (see [corpusQuery()]), most notably `expand`, a logical that decides if `query` and `vc` parameters are expanded to all of their combinations. It defaults to `TRUE`, if `query` and `vc` have different lengths, and to `FALSE` otherwise.
Marc Kupietz3f575282019-10-04 14:46:04 +02001387#' @export
Marc Kupietzad8d2ed2025-04-05 15:37:38 +02001388#'
1389#' @return A tibble, with each row containing the following result columns for query and vc combinations:
1390#' - **query**: the query string used for the frequency analysis.
1391#' - **totalResults**: absolute frequency of query matches in the vc.
1392#' - **vc**: virtual corpus used for the query.
1393#' - **webUIRequestUrl**: URL of the corresponding web UI request with respect to query and vc.
1394#' - **total**: total number of words in vc.
1395#' - **f**: relative frequency of query matches in the vc.
1396#' - **conf.low**: lower bound of the confidence interval for the relative frequency, given `conf.level`.
1397#' - **conf.high**: upper bound of the confidence interval for the relative frequency, given `conf.level`.
1398
Marc Kupietzd8851222025-05-01 10:57:19 +02001399setMethod(
1400 "frequencyQuery", "KorAPConnection",
Marc Kupietza3a8cd92026-09-08 07:59:04 +02001401 function(kco, query, vc = "", conf.level = 0.95, as.alternatives = FALSE,
1402 cacheAs = NULL, ...) {
1403 cacheRecord <- NULL
1404 if (!is.null(cacheAs)) {
1405 cacheAs <- cacheAsFileName(cacheAs)
1406 cacheRecord <- cacheAsRecord(environment(), list(...), kco)
1407 cached <- readCacheAs(cacheAs, kco, cacheRecord, "frequency query")
1408 if (!is.null(cached)) {
1409 return(cached)
1410 }
1411 }
1412
1413 result <- (if (as.alternatives) {
Marc Kupietzd8851222025-05-01 10:57:19 +02001414 corpusQuery(kco, query, vc, metadataOnly = TRUE, as.df = TRUE, ...) |>
Marc Kupietzea34b812025-06-25 15:49:00 +02001415 group_by(vc) |>
Marc Kupietz71d6e052019-11-22 18:42:10 +01001416 mutate(total = sum(totalResults))
Marc Kupietzd8851222025-05-01 10:57:19 +02001417 } else {
1418 corpusQuery(kco, query, vc, metadataOnly = TRUE, as.df = TRUE, ...) |>
1419 mutate(total = corpusStats(kco, vc = vc, as.df = TRUE)$tokens)
Marc Kupietzea34b812025-06-25 15:49:00 +02001420 }) |>
Marc Kupietz0c29cea2019-10-09 08:44:36 +02001421 ci(conf.level = conf.level)
Marc Kupietza3a8cd92026-09-08 07:59:04 +02001422
1423 if (!is.null(cacheAs)) {
1424 writeCacheAs(cacheAs, kco, cacheRecord, "frequency query", result)
1425 }
1426 result
Marc Kupietzd8851222025-05-01 10:57:19 +02001427 }
1428)
Marc Kupietz3f575282019-10-04 14:46:04 +02001429
Marc Kupietz38a9d682024-12-06 16:17:09 +01001430#' buildWebUIRequestUrlFromString
1431#'
1432#' @rdname KorAPQuery-class
1433#' @importFrom urltools url_encode
1434#' @export
1435buildWebUIRequestUrlFromString <- function(KorAPUrl,
Marc Kupietzd8851222025-05-01 10:57:19 +02001436 query,
1437 vc = "",
1438 ql = "poliqarp") {
Marc Kupietz38a9d682024-12-06 16:17:09 +01001439 if ("KorAPConnection" %in% class(KorAPUrl)) {
1440 KorAPUrl <- KorAPUrl@KorAPUrl
1441 }
1442
1443 request <-
1444 paste0(
Marc Kupietzd8851222025-05-01 10:57:19 +02001445 "?q=",
Marc Kupietz38a9d682024-12-06 16:17:09 +01001446 urltools::url_encode(enc2utf8(as.character(query))),
Marc Kupietzd8851222025-05-01 10:57:19 +02001447 ifelse(vc != "",
1448 paste0("&cq=", urltools::url_encode(enc2utf8(vc))),
1449 ""
1450 ),
1451 "&ql=",
Marc Kupietz38a9d682024-12-06 16:17:09 +01001452 ql
1453 )
1454 paste0(KorAPUrl, request)
1455}
Marc Kupietzdbd431a2021-08-29 12:17:45 +02001456
1457#' buildWebUIRequestUrl
1458#'
1459#' @rdname KorAPQuery-class
Marc Kupietzf9129592025-01-26 19:17:54 +01001460#' @importFrom httr2 url_parse
Marc Kupietzdbd431a2021-08-29 12:17:45 +02001461#' @export
1462buildWebUIRequestUrl <- function(kco,
Marc Kupietzd8851222025-05-01 10:57:19 +02001463 query = if (missing(KorAPUrl)) {
Marc Kupietzdbd431a2021-08-29 12:17:45 +02001464 stop("At least one of the parameters query and KorAPUrl must be specified.", call. = FALSE)
Marc Kupietzd8851222025-05-01 10:57:19 +02001465 } else {
1466 httr2::url_parse(KorAPUrl)$query$q
1467 },
Marc Kupietzf9129592025-01-26 19:17:54 +01001468 vc = if (missing(KorAPUrl)) "" else httr2::url_parse(KorAPUrl)$query$cq,
Marc Kupietzdbd431a2021-08-29 12:17:45 +02001469 KorAPUrl,
Marc Kupietzf9129592025-01-26 19:17:54 +01001470 ql = if (missing(KorAPUrl)) "poliqarp" else httr2::url_parse(KorAPUrl)$query$ql) {
Marc Kupietz38a9d682024-12-06 16:17:09 +01001471 buildWebUIRequestUrlFromString(kco@KorAPUrl, query, vc, ql)
Marc Kupietzdbd431a2021-08-29 12:17:45 +02001472}
1473
Marc Kupietzd8851222025-05-01 10:57:19 +02001474#' format()
Marc Kupietze95108e2019-09-18 13:23:58 +02001475#' @rdname KorAPQuery-class
1476#' @param x KorAPQuery object
1477#' @param ... further arguments passed to or from other methods
Marc Kupietzb73ca0f2025-01-28 20:45:01 +01001478#' @importFrom urltools param_get url_decode
Marc Kupietze95108e2019-09-18 13:23:58 +02001479#' @export
1480format.KorAPQuery <- function(x, ...) {
1481 cat("<KorAPQuery>\n")
1482 q <- x
Marc Kupietzd8851222025-05-01 10:57:19 +02001483 param <- urltools::param_get(q@request) |> lapply(urltools::url_decode)
Marc Kupietzb73ca0f2025-01-28 20:45:01 +01001484 cat(" Query: ", param$q, "\n")
1485 if (!is.null(param$cq) && param$cq != "") {
1486 cat(" Virtual corpus: ", param$cq, "\n")
1487 }
1488 if (!is.null(q@collectedMatches)) {
1489 cat("==============================================================================================================", "\n")
1490 print(summary(q@collectedMatches))
1491 cat("==============================================================================================================", "\n")
1492 }
1493 cat(" Total results: ", q@totalResults, "\n")
1494 cat(" Fetched results: ", q@nextStartIndex, "\n")
Marc Kupietza29f3d42025-07-18 10:14:43 +02001495 if (!is.null(q@collectedMatches) && "pos" %in% colnames(q@collectedMatches)) {
1496 successful_annotations <- sum(!is.na(q@collectedMatches$annotation_snippet))
1497 parsed_annotations <- sum(!is.na(q@collectedMatches$pos))
1498 cat(" Annotations: ", successful_annotations, " of ", nrow(q@collectedMatches), " matches")
1499 if (parsed_annotations > 0) {
1500 cat(" (", parsed_annotations, " with parsed linguistic data)")
1501 }
1502 cat("\n")
Marc Kupietze52b2952025-07-17 16:53:02 +02001503 }
Marc Kupietz62da2b52019-09-12 17:43:34 +02001504}
1505
Marc Kupietze95108e2019-09-18 13:23:58 +02001506#' show()
Marc Kupietz62da2b52019-09-12 17:43:34 +02001507#'
Marc Kupietze95108e2019-09-18 13:23:58 +02001508#' @rdname KorAPQuery-class
1509#' @param object KorAPQuery object
Marc Kupietz62da2b52019-09-12 17:43:34 +02001510#' @export
Marc Kupietze95108e2019-09-18 13:23:58 +02001511setMethod("show", "KorAPQuery", function(object) {
1512 format(object)
Marc Kupietzc643a122025-07-18 18:18:36 +02001513 invisible(object)
Marc Kupietze95108e2019-09-18 13:23:58 +02001514})