14b4ba83c35c34f4a1f3a69c9967f502ee2d6528

Author
Andrew Janke <andrew@apjanke.net>
Committer
Andrew Janke <andrew@apjanke.net>
Date

Message

Move urlencode/urldecode functions to core lib

Diff

This diff is truncated to protect this page.

  1diff --git a/lib/functions.zsh b/lib/functions.zsh
  2index 17f5f9cbf246ff0e6c58abfff6dff3ea227c99d8..5c1a5a283f6906ac0ca2c8aca47b69cd1fda169a 100644
  3--- a/lib/functions.zsh
  4+++ b/lib/functions.zsh
  5@@ -73,3 +73,137 @@ function env_default() {
  6     env | grep -q "^$1=" && return 0 
  7     export "$1=$2"       && return 3
  8 }
  9+
 10+
 11+# Required for $langinfo
 12+zmodload zsh/langinfo
 13+
 14+# URL-encode a string
 15+#
 16+# Encodes a string using RFC 2396 URL-encoding (%-escaped).
 17+# See: https://www.ietf.org/rfc/rfc2396.txt
 18+#
 19+# By default, reserved characters and unreserved "mark" characters are
 20+# not escaped by this function. This allows the common usage of passing
 21+# an entire URL in, and encoding just special characters in it, with 
 22+# the expectation that reserved and mark characters are used appropriately.
 23+# The -r and -m options turn on escaping of the reserved and mark characters,
 24+# respectively, which allows arbitrary strings to be fully escaped for
 25+# embedding inside URLs, where reserved characters might be misinterpreted.
 26+#
 27+# Prints the encoded string on stdout.
 28+# Returns nonzero if encoding failed.
 29+#
 30+# Usage:
 31+#  omz_urlencode [-r] [-m] <string>
 32+#  
 33+#    -r causes reserved characters (;/?:@&=+$,) to be escaped
 34+#
 35+#    -m causes "mark" characters (_.!~*''()-) to be escaped
 36+#
 37+#    -P causes spaces to be encoded as '%20' instead of '+'
 38+function omz_urlencode() {
 39+  emulate -L zsh
 40+  zparseopts -D -E -a opts r m P
 41+
 42+  local in_str=$1
 43+  local url_str=""
 44+  local spaces_as_plus
 45+  if [[ -z $opts[(r)-P] ]]; then spaces_as_plus=1; fi
 46+  local str="$in_str"
 47+
 48+  # URLs must use UTF-8 encoding; convert str to UTF-8 if required
 49+  local encoding=$langinfo[CODESET]
 50+  local safe_encodings
 51+  safe_encodings=(UTF-8 utf8 US-ASCII)
 52+  if [[ -z ${safe_encodings[(r)$encoding]} ]]; then
 53+    str=$(echo -E "$str" | iconv -f $encoding -t UTF-8)
 54+    if [[ $? != 0 ]]; then
 55+      echo "Error converting string from $encoding to UTF-8" >&2
 56+      return 1
 57+    fi
 58+  fi
 59+
 60+  # Use LC_CTYPE=C to process text byte-by-byte
 61+  local i byte ord LC_ALL=C
 62+  export LC_ALL
 63+  local reserved=';/?:@&=+$,'
 64+  local mark='_.!~*''()-'
 65+  local dont_escape="[A-Za-z0-9"
 66+  if [[ -z $opts[(r)-r] ]]; then
 67+    dont_escape+=$reserved
 68+  fi
 69+  # $mark must be last because of the "-"
 70+  if [[ -z $opts[(r)-m] ]]; then
 71+    dont_escape+=$mark
 72+  fi
 73+  dont_escape+="]"
 74+
 75+  # Implemented to use a single printf call and avoid subshells in the loop,
 76+  # for performance (primarily on Windows).
 77+  local url_str=""
 78+  for (( i = 1; i <= ${#str}; ++i )); do
 79+    byte="$str[i]"
 80+    if [[ "$byte" =~ "$dont_escape" ]]; then
 81+      url_str+="$byte"
 82+    else
 83+      if [[ "$byte" == " " && -n $spaces_as_plus ]]; then
 84+        url_str+="+"
 85+      else
 86+        ord=$(( [##16] #byte ))
 87+        url_str+="%$ord"
 88+      fi
 89+    fi
 90+  done
 91+  echo -E "$url_str"
 92+}
 93+
 94+# URL-decode a string
 95+#
 96+# Decodes a RFC 2396 URL-encoded (%-escaped) string.
 97+# This decodes the '+' and '%' escapes in the input string, and leaves 
 98+# other characters unchanged. Does not enforce that the input is a 
 99+# valid URL-encoded string. This is a convenience to allow callers to
100+# pass in a full URL or similar strings and decode them for human
101+# presentation.
102+#
103+# Outputs the encoded string on stdout.
104+# Returns nonzero if encoding failed.
105diff --git a/lib/termsupport.zsh b/lib/termsupport.zsh
106index 52622f5ab931d5fd99823063c5b837044d7e9c07..726cdce415057a8f3a12aa056b2b3eaba7d43e53 100644
107--- a/lib/termsupport.zsh
108+++ b/lib/termsupport.zsh
109@@ -59,44 +59,13 @@ preexec_functions+=(omz_termsupport_preexec)
110 
111 if [[ "$TERM_PROGRAM" == "Apple_Terminal" ]] && [[ -z "$INSIDE_EMACS" ]]; then
112 
113-  # URL-encodes a string
114-  # Outputs the encoded string on stdout
115-  # Returns nonzero if encoding failed
116-  function _omz_urlencode() {
117-    local str=$1
118-    local url_str=""
119-
120-    # URLs must use UTF-8 encoding; convert if required
121-    local encoding=${LC_CTYPE/*./}
122-    if [[ -n $encoding && $encoding != UTF-8 && $encoding != utf8 ]]; then
123-      str=$(echo $str | iconv -f $encoding -t UTF-8)
124-      if [[ $? != 0 ]]; then
125-        echo "Error converting string from $encoding to UTF-8" >&2
126-        return 1
127-      fi
128-    fi
129-
130-    # Use LC_CTYPE=C to process text byte-by-byte
131-    local i ch hexch LC_CTYPE=C
132-    for ((i = 1; i <= ${#str}; ++i)); do
133-      ch="$str[i]"
134-      if [[ "$ch" =~ [/._~A-Za-z0-9-] ]]; then
135-        url_str+="$ch"
136-      else
137-        hexch=$(printf "%02X" "'$ch")
138-        url_str+="%$hexch"
139-      fi
140-    done
141-    echo $url_str
142-  }
143-
144   # Emits the control sequence to notify Terminal.app of the cwd
145   function update_terminalapp_cwd() {
146     # Identify the directory using a "file:" scheme URL, including
147     # the host name to disambiguate local vs. remote paths.
148 
149     # Percent-encode the pathname.
150-    local URL_PATH=$(_omz_urlencode $PWD)
151+    local URL_PATH=$(omz_urlencode -P $PWD)
152     [[ $? != 0 ]] && return 1
153     local PWD_URL="file://$HOST$URL_PATH"
154     # Undocumented Terminal.app-specific control sequence