input/csv: Skip leading UTF-8 BOM in the input stream

author Gerhard Sittig <redacted>

Mon, 5 Jun 2017 16:35:22 +0000 (18:35 +0200)

committer Uwe Hermann <redacted>

Tue, 6 Jun 2017 21:28:05 +0000 (23:28 +0200)
author Gerhard Sittig <redacted>
Mon, 5 Jun 2017 16:35:22 +0000 (18:35 +0200)
committer Uwe Hermann <redacted>
Tue, 6 Jun 2017 21:28:05 +0000 (23:28 +0200)
diff --git a/src/input/csv.c b/src/input/csv.c

index 013d7a476b3338b6b14e993e8989b78019c371b7..baf37188ad6427ffd903ab31b62197e651dd45b4 100644 (file)
--- a/src/input/csv.c
+++ b/src/input/csv.c
@@ -613,6 +613,27 @@ out:
         return ret;
  }
  
+/*
+ * Gets called from initial_receive(), which runs until the end-of-line
+ * encoding of the input stream could get determined. Assumes that this
+ * routine receives enough buffered initial input data to either see the
+ * BOM when there is one, or that no BOM will follow when a text line
+ * termination sequence was seen. Silently drops the UTF-8 BOM sequence
+ * from the input buffer if one was seen. Does not care to protect
+ * against multiple execution or dropping the BOM multiple times --
+ * there should be at most one in the input stream.
+ */
+static void initial_bom_check(const struct sr_input *in)
+{
+       static const char *utf8_bom = "\xef\xbb\xbf";
+
+       if (in->buf->len < strlen(utf8_bom))
+               return;
+       if (strncmp(in->buf->str, utf8_bom, strlen(utf8_bom)) != 0)
+               return;
+       g_string_erase(in->buf, 0, strlen(utf8_bom));
+}
+
  static int initial_receive(const struct sr_input *in)
  {
         struct context *inc;
@@ -621,6 +642,8 @@ static int initial_receive(const struct sr_input *in)
         char *p;
         const char *termination;
  
+       initial_bom_check(in);
+
         inc = in->priv;
  
         termination = get_line_termination(in->buf);
author	Gerhard Sittig <redacted>
	Mon, 5 Jun 2017 16:35:22 +0000 (18:35 +0200)
committer	Uwe Hermann <redacted>
	Tue, 6 Jun 2017 21:28:05 +0000 (23:28 +0200)