csv - awk 根据单列的唯一值组合其他列的唯一值
问题描述
我的输入文件看起来像
Item1,200,a,four,five,six,seven,eight1,nine1
Item2,500,b,four,five,six,seven,eight2,nine2
Item3,900,c,four,five,six,seven,eight3,nine3
Item2,800,d,four,five,six,seven,eight4,nine4
Item1,,e,four,five,six,seven,eight5,nine5
基于第一列的唯一值,我想组合所有其他列的唯一值。到目前为止我尝试的是:
awk -F, '{
a[$1]=a[$1]?a[$1]"_"$2:$2;
b[$1]=b[$1]?b[$1]"_"$3:$3;
c[$1]=c[$1]?c[$1]"_"$4:$4;
d[$1]=d[$1]?d[$1]"_"$5:$5;
e[$1]=e[$1]?e[$1]"_"$6:$6;
f[$1]=f[$1]?f[$1]"_"$7:$7;
g[$1]=g[$1]?g[$1]"_"$8:$8;
h[$1]=h[$1]?h[$1]"_"$9:$9;
}END{for (i in a)print i, a[i], b[i], c[i], d[i], e[i], f[i], g[i], h[i];}' OFS=, input.txt
上面的输出是:
Item3,900,c,four,five,six,seven,eight3,nine3
Item1,200_,a_e,four_four,five_five,six_six,seven_seven,eight1_eight5,nine1_nine5
Item2,500_800,b_d,four_four,five_five,six_six,seven_seven,eight2_eight4,nine2_nine4
但我期待的是:
Item3,900,c,four,five,six,seven,eight3,nine3
Item1,200,a_e,four,five,six,seven,eight1_eight5,nine1_nine5
Item2,500_800,b_d,four,five,six,seven,eight2_eight4,nine2_nine4
我正在寻求一些帮助:
- 如何在组合值时只取唯一值?
- 每当存在空白值时,在组合时不应在末尾附加分隔符(在我上面的例子中为下划线)?
- 如何根据第 1 列的值对输出进行排序?
非常感谢你的帮助。
解决方案
任何awk
加号sort
:
$ cat tst.awk
BEGIN { FS=OFS="," }
{
key = $1
keys[key]
for (i=2; i<=NF; i++) {
if ( ($i ~ /[^[:space:]]/) && (!seen[key,i,$i]++) ) {
idx = key FS i
vals[idx] = (idx in vals ? vals[idx] "_" : "") $i
}
}
}
END {
for (key in keys) {
printf "%s%s", key, OFS
for (i=2; i<=NF; i++) {
idx = key FS i
printf "%s%s", vals[idx], (i<NF ? OFS : ORS)
}
}
}
.
$ awk -f tst.awk file | sort -t, -k1,1
Item1,200,a_e,four,five,six,seven,eight1_eight5,nine1_nine5
Item2,500_800,b_d,four,five,six,seven,eight2_eight4,nine2_nine4
Item3,900,c,four,five,six,seven,eight3,nine3
或者使用 GNUawk
来处理数组(参见https://www.gnu.org/software/gawk/manual/gawk.html#Multidimensional和https://www.gnu.org/software/gawk/manual/gawk.html #Arrays-of-Arrays了解两者之间的区别)和sorted_in
(参见https://www.gnu.org/software/gawk/manual/gawk.html#Controlling-Array-Traversal和https://www.gnu。 org/software/gawk/manual/gawk.html#Controlling-Scanning):
$ cat tst.awk
BEGIN { FS=OFS="," }
{
for ( i=2; i<=NF; i++ ) {
vals[$1][i][$i]
}
}
END {
PROCINFO["sorted_in"] = "@ind_str_asc"
for ( key in vals ) {
printf "%s%s", key, OFS
for ( i=2; i<=NF; i++ ) {
sep = ""
for ( val in vals[key][i] ) {
if ( val ~ /[^[:space:]]/ ) {
printf "%s%s", sep, val
sep = "_"
}
}
printf "%s", (i<NF ? OFS : ORS)
}
}
}
.
$ awk -f tst.awk file
Item1,200,a_e,four,five,six,seven,eight1_eight5,nine1_nine5
Item2,500_800,b_d,four,five,six,seven,eight2_eight4,nine2_nine4
Item3,900,c,four,five,six,seven,eight3,nine3