test_that("automatic schema detection recognizes English and Persian columns", { en <- data.frame(Value = 10.4, Indicator = "unemployment rate", Country = "Spain", Year = 2025, Unit = "percent", Source = "Example NSO", check.names = FALSE) s1 <- detect_stat_schema(en) expect_equal(s1$mapping$value, "Value") expect_equal(s1$mapping$geo, "Country") expect_equal(s1$mapping$time, "Year") fa <- data.frame("مقدار" = "۱۲٫۵", "عنوان شاخص" = "نرخ بیکاری", "کشور" = "ایران", "سال" = "۱۴۰۳", "واحد" = "درصد", check.names = FALSE) s2 <- detect_stat_schema(fa) expect_equal(s2$mapping$value, "مقدار") expect_equal(s2$mapping$indicator, "عنوان شاخص") expect_equal(s2$mapping$geo, "کشور") expect_equal(s2$mapping$time, "سال") }) test_that("generic importer normalizes Persian digits and units", { fa <- data.frame("مقدار" = "۱۲٫۵", "عنوان شاخص" = "نرخ بیکاری", "کشور" = "ایران", "سال" = "۱۴۰۳", "واحد" = "درصد", check.names = FALSE) x <- read_official_stats(fa, provider = "Statistical Center of Iran") expect_s3_class(x, "stat_provider_data") expect_equal(x$.value, 12.5) expect_equal(x$.time, "1403") expect_equal(x$.unit, "percent") ref <- as_stat_reference(x) expect_equal(ref$source, "Statistical Center of Iran") }) test_that("easy audit accepts ordinary data and plain text", { d <- data.frame(value = 10.4, indicator = "unemployment rate", country = "Spain", year = 2025, unit = "percent", source = "Example NSO", stringsAsFactors = FALSE) a <- audit_stats(d, "The reported value is 10.4%.") expect_s3_class(a, "stat_fidelity_audit") expect_equal(a$components$numerical$score, 1) expect_equal(a$components$unit$score, 1) }) test_that("generic provider profiles can carry reusable mappings", { register_official_provider( "demo_nso", "Demo National Statistical Office", mapping = list(value = "V", indicator = "I", geo = "G", time = "T", unit = "U"), host_patterns = "data.demo.example", overwrite = TRUE ) d <- data.frame(V = 3.2, I = "test rate", G = "Exampleland", T = 2025, U = "percent") x <- read_official_stats(d, provider = "demo_nso") expect_equal(attr(x, "provider"), "Demo National Statistical Office") expect_equal(x$.value, 3.2) expect_equal(detect_stat_provider("https://data.demo.example/table.csv"), "demo_nso") }) test_that("provider detection includes WHO and Iranian organisations", { expect_equal(detect_stat_provider("https://data.who.int/indicators"), "who") expect_equal(detect_stat_provider("https://www.amar.org.ir/data.csv"), "sci_iran") x <- official_stat_providers(include_custom = FALSE) expect_true(all(c("who", "sci_iran", "cbi_iran", "mohme_iran") %in% x$id)) expect_true(all(x$generic_import[x$id %in% c("who", "sci_iran")])) }) test_that("plain-text claim extraction accepts Persian digits", { x <- extract_stat_claims("مقدار گزارش شده ۱۲٫۵٪ است.", hints = list(indicator = "نرخ", geo = "ایران", time = "1403")) expect_equal(length(x), 1) expect_equal(x[[1]]$value, 12.5) expect_equal(x[[1]]$unit, "percent") })