【问题标题】:How do I remove white space from between letters and not numbers?如何从字母而不是数字之间删除空格?
【发布时间】:2020-12-26 08:26:27
【问题描述】:

我有想要转换为表格的字符串。每行中的标识符可以有空格,我需要删除它们而不删除数字之间的空格。是否可以使用正则表达式来实现这一点?

例如,数据如下所示:

A B C 5.65 7.8 
DC 5.65 7.8
D AB   7.9  12.2
D AB C  7.9  1.2
A BC 13.88 2.4
AB C  7.9  12.2

我想解决这个问题:

ABC 5.65 7.8 
DC 5.65 7.8
DAB   7.9  12.2
DABC  7.9  1.2
ABC 13.88 2.4
ABC  7.9  12.2

编辑:根据要求,这是我收到它的数据类型和形式的示例。这有 16 行,每行有 6 列数据,但第一列是字母标识符。

 # Data as I receive it.

data <- c("A", "a", "2.07", "2.35", "39.00", "82.20", "8.8", "3.80", 
           "B", "2.26", "2.25", "40.00", "80.80", "8.1", "1.86", "D", 
           "Et", "2.07", "2.22", "41.00", "83.80", "8.8", "3.87", "F", 
    "2.05", "2.15", "43.00", "82.20", "8.4", "3.11", "Bc", "2.08", 
    "2.12", "48.00", "82.60", "8.3", "2.47", "Gf", "H", "I", 
    "2.08", "2.10", "46.00", "82.20", "8.1", "2.90", "J", "K", 
    "1.95", "2.08", "38.00", "83.40", "8.7", "1.63", "L", "M", 
    "1.89", "2.07", "45.00", "83.80", "9.0", "1.84", "N", "2.06", 
    "2.05", "41.00", "80.60", "9.0", "4.09", "O", "P", "1.86", 
    "2.04", "48.00", "81.60", "8.6", "2.60", "Qst", "R", "1.95", 
    "2.03", "44.00", "82.80", "8.8", "1.40", "S", "2.03", "2.02", 
    "40.00", "81.40", "8.2", "1.74", "T", "1.95", "2.01", "43.00", 
    "81.80", "9.0", "2.30", "Unh", "1.96", "2.00", "44.00", "82.60", 
    "9.2", "2.40", "V", "W", "C", "1.98", "1.97", "40.00", 
    "82.00", "8.1", "1.15", "Yu", "1.90", "1.96", "41.00", "82.80", 
    "9.6", "2.08", "Z", "a", "bi", "1.90", "1.95", "42.00", 
    "84.20", "9.6", "1.69")
    
# Required format

data2 <- c("Aa", "2.07", "2.35", "39.00", "82.20", "8.8", "3.80", 
          "B", "2.26", "2.25", "40.00", "80.80", "8.1", "1.86", 
          "DEt", "2.07", "2.22", "41.00", "83.80", "8.8", "3.87", "F", 
          "2.05", "2.15", "43.00", "82.20", "8.4", "3.11", "Bc", "2.08", 
          "2.12", "48.00", "82.60", "8.3", "2.47", "GfHI", 
          "2.08", "2.10", "46.00", "82.20", "8.1", "2.90", "JK", 
          "1.95", "2.08", "38.00", "83.40", "8.7", "1.63", "LM", 
          "1.89", "2.07", "45.00", "83.80", "9.0", "1.84", "N", "2.06", 
          "2.05", "41.00", "80.60", "9.0", "4.09", "OP", "1.86", 
          "2.04", "48.00", "81.60", "8.6", "2.60", "QstR", "1.95", 
          "2.03", "44.00", "82.80", "8.8", "1.40", "S", "2.03", "2.02", 
          "40.00", "81.40", "8.2", "1.74", "T", "1.95", "2.01", "43.00", 
          "81.80", "9.0", "2.30", "Unh", "1.96", "2.00", "44.00", "82.60", 
          "9.2", "2.40", "VWC", "1.98", "1.97", "40.00", 
          "82.00", "8.1", "1.15", "Yu", "1.90", "1.96", "41.00", "82.80", 
          "9.6", "2.08", "Zabi", "1.90", "1.95", "42.00", 
          "84.20", "9.6", "1.69")

df <- data.frame(matrix(data2, ncol=7, byrow=T))

【问题讨论】:

标签: r removing-whitespace


【解决方案1】:

要在 R 环境中按照您的要求执行操作,一种方法是将向量转换为字符串,对字符串应用正则表达式过滤器,然后将字符串转换回向量。

请参阅下面的详细信息,希望这会为您指明正确的方向。

解决方案

data <- c("A", "a", "2.07", "2.35", "39.00", "82.20", "8.8", "3.80",
          "B", "2.26", "2.25", "40.00", "80.80", "8.1", "1.86", "D",
          "Et", "2.07", "2.22", "41.00", "83.80", "8.8", "3.87", "F",
          "2.05", "2.15", "43.00", "82.20", "8.4", "3.11", "Bc", "2.08",
          "2.12", "48.00", "82.60", "8.3", "2.47", "Gf", "H", "I",
          "2.08", "2.10", "46.00", "82.20", "8.1", "2.90", "J", "K",
          "1.95", "2.08", "38.00", "83.40", "8.7", "1.63", "L", "M",
          "1.89", "2.07", "45.00", "83.80", "9.0", "1.84", "N", "2.06",
          "2.05", "41.00", "80.60", "9.0", "4.09", "O", "P", "1.86",
          "2.04", "48.00", "81.60", "8.6", "2.60", "Qst", "R", "1.95",
          "2.03", "44.00", "82.80", "8.8", "1.40", "S", "2.03", "2.02",
          "40.00", "81.40", "8.2", "1.74", "T", "1.95", "2.01", "43.00",
          "81.80", "9.0", "2.30", "Unh", "1.96", "2.00", "44.00", "82.60",
          "9.2", "2.40", "V", "W", "C", "1.98", "1.97", "40.00",
          "82.00", "8.1", "1.15", "Yu", "1.90", "1.96", "41.00", "82.80",
          "9.6", "2.08", "Z", "a", "bi", "1.90", "1.95", "42.00",
          "84.20", "9.6", "1.69")

# Use stringi base regular expression engine
require(stringi)

# Convert the vector data to be a string sequence - so we can manipulate as text
data1 <- toString(data)

# Now we can apply the regular expression substitution to the data (formatted as a string...
# Here we do a:
#
#     (?<!\d) - Negative look behind to prevent a digit.
#     ,  - A literal combination of quotes, comma and space. We drop the ", " in conversion to string...
#     (?!\d) - Negative look ahead to prevent a digit.
#
data3 = stri_replace_all_regex(str = data1, pattern = '(?<!\\d), (?!\\d)', replacement = '')
# OK, check the string data...
data3

# Now we convert the string back to be a vector...
newData = strsplit(data3, " ")[[1]]
newData 

# Now we convert to a dataframe...
df <- data.frame(matrix(newData, ncol=7, byrow=T))
df
# Done

输出

> data <- c("A", "a", "2.07", "2.35", "39.00", "82.20", "8.8", "3.80",
+           "B", "2.26", "2.25", "40.00", "80.80", "8.1", "1.86", "D",
+           "Et", "2.07", "2.22", "41.00", "83.80", "8.8", "3.87", "F",
+           "2.05", "2.15", "43.00", "82.20", "8.4", "3.11", "Bc", "2.08",
+           "2.12", "48.00", "82.60", "8.3", "2.47", "Gf", "H", "I",
+           "2.08", "2.10", "46.00", "82.20", "8.1", "2.90", "J", "K",
+           "1.95", "2.08", "38.00", "83.40", "8.7", "1.63", "L", "M",
+           "1.89", "2.07", "45.00", "83.80", "9.0", "1.84", "N", "2.06",
+           "2.05", "41.00", "80.60", "9.0", "4.09", "O", "P", "1.86",
+           "2.04", "48.00", "81.60", "8.6", "2.60", "Qst", "R", "1.95",
+           "2.03", "44.00", "82.80", "8.8", "1.40", "S", "2.03", "2.02",
+           "40.00", "81.40", "8.2", "1.74", "T", "1.95", "2.01", "43.00",
+           "81.80", "9.0", "2.30", "Unh", "1.96", "2.00", "44.00", "82.60",
+           "9.2", "2.40", "V", "W", "C", "1.98", "1.97", "40.00",
+           "82.00", "8.1", "1.15", "Yu", "1.90", "1.96", "41.00", "82.80",
+           "9.6", "2.08", "Z", "a", "bi", "1.90", "1.95", "42.00",
+           "84.20", "9.6", "1.69")
> 
> # Use stringi base regular expression engine
> require(stringi)
> 
> # Convert the vector data to be a string sequence - so we can manipulate as text
> data1 <- toString(data)
> 
> # Now we can apply the regular expression substitution to the data (formatted as a string...
> # Here we do a:
> #
> #     (?<!\d) - Negative look behind to prevent a digit.
> #     ,  - A literal combination of quotes, comma and space. We drop the ", " in conversion to string...
> #     (?!\d) - Negative look ahead to prevent a digit.
> #
> data3 = stri_replace_all_regex(str = data1, pattern = '(?<!\\d), (?!\\d)', replacement = '')
> # OK, check the string data...
> data3
[1] "Aa, 2.07, 2.35, 39.00, 82.20, 8.8, 3.80, B, 2.26, 2.25, 40.00, 80.80, 8.1, 1.86, DEt, 2.07, 2.22, 41.00, 83.80, 8.8, 3.87, F, 2.05, 2.15, 43.00, 82.20, 8.4, 3.11, Bc, 2.08, 2.12, 48.00, 82.60, 8.3, 2.47, GfHI, 2.08, 2.10, 46.00, 82.20, 8.1, 2.90, JK, 1.95, 2.08, 38.00, 83.40, 8.7, 1.63, LM, 1.89, 2.07, 45.00, 83.80, 9.0, 1.84, N, 2.06, 2.05, 41.00, 80.60, 9.0, 4.09, OP, 1.86, 2.04, 48.00, 81.60, 8.6, 2.60, QstR, 1.95, 2.03, 44.00, 82.80, 8.8, 1.40, S, 2.03, 2.02, 40.00, 81.40, 8.2, 1.74, T, 1.95, 2.01, 43.00, 81.80, 9.0, 2.30, Unh, 1.96, 2.00, 44.00, 82.60, 9.2, 2.40, VWC, 1.98, 1.97, 40.00, 82.00, 8.1, 1.15, Yu, 1.90, 1.96, 41.00, 82.80, 9.6, 2.08, Zabi, 1.90, 1.95, 42.00, 84.20, 9.6, 1.69"
> 
> # Now we convert the string back to be a vector...
> newData = strsplit(data3, " ")[[1]]
> newData 
  [1] "Aa,"    "2.07,"  "2.35,"  "39.00," "82.20," "8.8,"   "3.80,"  "B,"     "2.26,"  "2.25,"  "40.00," "80.80,"
 [13] "8.1,"   "1.86,"  "DEt,"   "2.07,"  "2.22,"  "41.00," "83.80," "8.8,"   "3.87,"  "F,"     "2.05,"  "2.15," 
 [25] "43.00," "82.20," "8.4,"   "3.11,"  "Bc,"    "2.08,"  "2.12,"  "48.00," "82.60," "8.3,"   "2.47,"  "GfHI," 
 [37] "2.08,"  "2.10,"  "46.00," "82.20," "8.1,"   "2.90,"  "JK,"    "1.95,"  "2.08,"  "38.00," "83.40," "8.7,"  
 [49] "1.63,"  "LM,"    "1.89,"  "2.07,"  "45.00," "83.80," "9.0,"   "1.84,"  "N,"     "2.06,"  "2.05,"  "41.00,"
 [61] "80.60," "9.0,"   "4.09,"  "OP,"    "1.86,"  "2.04,"  "48.00," "81.60," "8.6,"   "2.60,"  "QstR,"  "1.95," 
 [73] "2.03,"  "44.00," "82.80," "8.8,"   "1.40,"  "S,"     "2.03,"  "2.02,"  "40.00," "81.40," "8.2,"   "1.74," 
 [85] "T,"     "1.95,"  "2.01,"  "43.00," "81.80," "9.0,"   "2.30,"  "Unh,"   "1.96,"  "2.00,"  "44.00," "82.60,"
 [97] "9.2,"   "2.40,"  "VWC,"   "1.98,"  "1.97,"  "40.00," "82.00," "8.1,"   "1.15,"  "Yu,"    "1.90,"  "1.96," 
[109] "41.00," "82.80," "9.6,"   "2.08,"  "Zabi,"  "1.90,"  "1.95,"  "42.00," "84.20," "9.6,"   "1.69"  
> 
> # Now we convert to a dataframe...
> df <- data.frame(matrix(newData, ncol=7, byrow=T))
> df
      X1    X2    X3     X4     X5   X6    X7
1    Aa, 2.07, 2.35, 39.00, 82.20, 8.8, 3.80,
2     B, 2.26, 2.25, 40.00, 80.80, 8.1, 1.86,
3   DEt, 2.07, 2.22, 41.00, 83.80, 8.8, 3.87,
4     F, 2.05, 2.15, 43.00, 82.20, 8.4, 3.11,
5    Bc, 2.08, 2.12, 48.00, 82.60, 8.3, 2.47,
6  GfHI, 2.08, 2.10, 46.00, 82.20, 8.1, 2.90,
7    JK, 1.95, 2.08, 38.00, 83.40, 8.7, 1.63,
8    LM, 1.89, 2.07, 45.00, 83.80, 9.0, 1.84,
9     N, 2.06, 2.05, 41.00, 80.60, 9.0, 4.09,
10   OP, 1.86, 2.04, 48.00, 81.60, 8.6, 2.60,
11 QstR, 1.95, 2.03, 44.00, 82.80, 8.8, 1.40,
12    S, 2.03, 2.02, 40.00, 81.40, 8.2, 1.74,
13    T, 1.95, 2.01, 43.00, 81.80, 9.0, 2.30,
14  Unh, 1.96, 2.00, 44.00, 82.60, 9.2, 2.40,
15  VWC, 1.98, 1.97, 40.00, 82.00, 8.1, 1.15,
16   Yu, 1.90, 1.96, 41.00, 82.80, 9.6, 2.08,
17 Zabi, 1.90, 1.95, 42.00, 84.20, 9.6,  1.69
> # Done

【讨论】:

  • 感谢您的帮助。是否可以在 r 环境中执行此操作?考虑到我现有的工作流程,这更可取。
  • @NicholasGeorge 我更新了在 R 环境中做所有事情的答案。
  • @nicholas-george 好的,我更新了答案以反映您对仅 R 环境的解决方案的需求。感谢您发布此内容,它迫使我去复习我的 R 正则表达式技能和缺乏的技能。我希望这能为您指明正确的方向。小心并保持安全。
猜你喜欢
  • 1970-01-01
  • 2019-04-24
  • 2018-03-12
  • 1970-01-01
  • 1970-01-01
  • 1970-01-01
  • 1970-01-01
  • 2023-03-04
  • 1970-01-01
相关资源
最近更新 更多