URL decoding
This task (the reverse of URL encoding and distinct from URL parser) is to provide a function or mechanism to convert an URL-encoded string into its original unencoded form.

You are encouraged to solve this task according to the task description, using any language you may know.
- Test cases
- The encoded string "
http%3A%2F%2Ffoo%20bar%2F" should be reverted to the unencoded form "http://foo bar/".
- The encoded string "
google.com/search?q=%60Abdu%27l-Bah%C3%A1" should revert to the unencoded form "google.com/search?q=`Abdu'l-Bahá".
- The encoded string "
%25%32%35" should revert to the unencoded form "%25" and not "%".
F url_decode(s)
V r = ‘’
V i = 0
L i < s.len
I s[i] == ‘%’
[Byte] b
L i < s.len & s[i] == ‘%’
i++
b.append(Int(s[i.+2], radix' 16))
i += 2
r ‘’= b.decode(‘utf-8’)
E
r ‘’= s[i]
i++
R r
print(url_decode(‘http%3A%2F%2Ffoo%20bar%2F’))
print(url_decode(‘https://ru.wikipedia.org/wiki/%D0%A2%D1%80%D0%B0%D0%BD%D1%81%D0%BF%D0%B0%D0%B9%D0%BB%D0%B5%D1%80’))- Output:
http://foo bar/ https://ru.wikipedia.org/wiki/Транспайлер
REPORT Z_DECODE_URL.
DATA: lv_encoded_url TYPE string VALUE 'http%3A%2F%2Ffoo%20bar%2F',
lv_decoded_url TYPE string.
CALL METHOD CL_HTTP_UTILITY=>UNESCAPE_URL
EXPORTING
ESCAPED = lv_encoded_url
RECEIVING
UNESCAPED = lv_decoded_url.
WRITE: 'Encoded URL: ', lv_encoded_url, /, 'Decoded URL: ', lv_decoded_url.
PROC Append(CHAR ARRAY s CHAR c)
s(0)==+1
s(s(0))=c
RETURN
CHAR FUNC GetCharFromHex(CHAR c1,c2)
CHAR ARRAY hex=['0 '1 '2 '3 '4 '5 '6 '7 '8 '9 'A 'B 'C 'D 'E 'F]
BYTE i,res
res=0
FOR i=0 TO 15
DO
IF c1=hex(i) THEN res==+i LSH 4 FI
IF c2=hex(i) THEN res==+i FI
OD
RETURN (res)
PROC Decode(CHAR ARRAY in,out)
BYTE i
CHAR c
out(0)=0
i=1
WHILE i<=in(0)
DO
c=in(i)
i==+1
IF c='+ THEN
Append(out,' )
ELSEIF c='% THEN
c=GetCharFromHex(in(i),in(i+1))
i==+2
Append(out,c)
ELSE
Append(out,c)
FI
OD
RETURN
PROC PrintInv(CHAR ARRAY a)
BYTE i
IF a(0)>0 THEN
FOR i=1 TO a(0)
DO
Put(a(i)%$80)
OD
FI
RETURN
PROC Test(CHAR ARRAY in)
CHAR ARRAY out(256)
PrintInv("input ")
PrintF(" %S%E",in)
Decode(in,out)
PrintInv("decoded")
PrintF(" %S%E%E",out)
RETURN
PROC Main()
Test("http%3A%2F%2Ffoo%20bar%2F")
Test("http%3A%2F%2Ffoo+bar%2F*_-.html")
RETURN- Output:
Screenshot from Atari 8-bit computer
input http%3A%2F%2Ffoo%20bar%2F decoded http://foo bar/ input http%3A%2F%2Ffoo+bar%2F*_-.html decoded http://foo bar/*_-.html
with AWS.URL;
with Ada.Text_IO; use Ada.Text_IO;
procedure Decode is
Encoded : constant String := "http%3A%2F%2Ffoo%20bar%2F";
begin
Put_Line (AWS.URL.Decode (Encoded));
end Decode;
Without external libraries:
package URL is
function Decode (URL : in String) return String;
end URL;
package body URL is
function Decode (URL : in String) return String is
Buffer : String (1 .. URL'Length);
Filled : Natural := 0;
Position : Positive := URL'First;
begin
while Position in URL'Range loop
Filled := Filled + 1;
case URL (Position) is
when '+' =>
Buffer (Filled) := ' ';
Position := Position + 1;
when '%' =>
Buffer (Filled) :=
Character'Val
(Natural'Value
("16#" & URL (Position + 1 .. Position + 2) & "#"));
Position := Position + 3;
when others =>
Buffer (Filled) := URL (Position);
Position := Position + 1;
end case;
end loop;
return Buffer (1 .. Filled);
end Decode;
end URL;
with Ada.Command_Line,
Ada.Text_IO;
with URL;
procedure Test_URL_Decode is
use Ada.Command_Line, Ada.Text_IO;
begin
if Argument_Count = 0 then
Put_Line (URL.Decode ("http%3A%2F%2Ffoo%20bar%2F"));
else
for I in 1 .. Argument_Count loop
Put_Line (URL.Decode (Argument (I)));
end loop;
end if;
end Test_URL_Decode;
Note: See the discussion page re displaying the results (which is mostly focused on the AWK solution but will apply elsewhere too).
For the second task example, this outputs the encoded UTF-8 characters, what you see depends on what you look at it with...
# returns c decoded as a hex digit #
PROC hex value = ( CHAR c )INT: IF c >= "0" AND c <= "9" THEN ABS c - ABS "0"
ELIF c >= "A" AND c <= "F" THEN 10 + ( ABS c - ABS "A" )
ELSE 10 + ( ABS c - ABS "a" )
FI;
# returns the URL encoded string decoded - minimal error handling #
PROC url decode = ( STRING encoded )STRING:
BEGIN
[ LWB encoded : UPB encoded ]CHAR result;
INT result pos := LWB encoded;
INT pos := LWB encoded;
INT max pos := UPB encoded;
INT max encoded := max pos - 3;
WHILE pos <= UPB encoded
DO
IF encoded[ pos ] /= "%" AND pos <= max encoded
THEN
# not a coded character #
result[ result pos ] := encoded[ pos ];
pos +:= 1
ELSE
# have an encoded character #
result[ result pos ] := REPR ( ( 16 * hex value( encoded[ pos + 1 ] ) )
+ hex value( encoded[ pos + 2 ] )
);
pos +:= 3
FI;
result pos +:= 1
OD;
result[ LWB result : result pos - 1 ]
END # url decode # ;
# test the url decode procedure #
print( ( url decode( "http%3A%2F%2Ffoo%20bar%2F" ), newline ) );
print( ( url decode( "google.com/search?q=%60Abdu%27l-Bah%C3%A1" ), newline ) )- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
EncodingUtil.urlDecode('http%3A%2F%2Ffoo%20bar%2F', 'UTF-8');
EncodingUtil.urlDecode('google.com/search?q=%60Abdu%27l-Bah%C3%A1', 'UTF-8');http://foo bar/ google.com/search?q=`Abdu'l-Bahá
AST URL decode "google.com/search?q=%60Abdu%27l-Bah%C3%A1"
print decode.url "http%3A%2F%2Ffoo%20bar%2F"
print decode.url "google.com/search?q=%60Abdu%27l-Bah%C3%A1"
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
UriDecode(Uri) {
LoopOffset := 0
VarLength := 0
VarSetCapacity(Var, StrPut(Uri, "UTF-8"), 0)
Loop Parse, Uri
{
If (A_Index < LoopOffset) {
Continue
}
If (A_LoopField = Chr(37)) {
Number := "0x" . SubStr(Uri, A_Index + 1, 2)
LoopOffset := A_Index + 3
}
Else {
Number := Ord(A_LoopField)
}
NumPut(Number, Var, VarLength++, "UChar")
}
Return StrGet(&Var, VarLength, "UTF-8")
}
MsgBox % UriDecode("http%3A%2F%2Ffoo%20bar%2F")
MsgBox % UriDecode("google.com/search?q=%60Abdu%27l-Bah%C3%A1")
MsgBox % UriDecode("%25%32%35")
# syntax:
awk '
BEGIN {
str = "http%3A%2F%2Ffoo%20bar%2F" # "http://foo bar/"
printf("%s\n",str)
len=length(str)
for (i=1;i<=len;i++) {
if ( substr(str,i,1) == "%") {
L = substr(str,1,i-1) # chars to left of "%"
M = substr(str,i+1,2) # 2 chars to right of "%"
R = substr(str,i+3) # chars to right of "%xx"
str = sprintf("%s%c%s",L,hex2dec(M),R)
}
}
printf("%s\n",str)
exit(0)
}
function hex2dec(s, num) {
num = index("0123456789ABCDEF",toupper(substr(s,length(s)))) - 1
sub(/.$/,"",s)
return num + (length(s) ? 16*hex2dec(s) : 0)
} '
- Output:
http%3A%2F%2Ffoo%20bar%2F http://foo bar/
OR:
LC_ALL=C
echo "http%3A%2F%2Ffoo%20bar%2F" | gawk -vRS='%[[:xdigit:]]{2}' '
RT {RT = sprintf("%c",strtonum("0x" substr(RT, 2)))}
{gsub(/+/," ");printf "%s", $0 RT}'
- Output:
http://foo bar/
FUNCTION Url_Decode$(url$)
LOCAL result$
SPLIT url$ BY "%" TO item$ SIZE total
FOR x = 1 TO total-1
result$ = result$ & CHR$(DEC(LEFT$(item$[x], 2))) & MID$(item$[x], 3)
NEXT
RETURN item$[0] & result$
END FUNCTION
PRINT Url_Decode$("http%3A%2F%2Ffoo%20bar%2F")
PRINT Url_Decode$("google.com/search?q=%60Abdu%27l-Bah%C3%A1")
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
See #UNIX Shell
Unicode
This solution parses all hex values in one go, and thus is technically faster than the ANSI one.
@echo off
setlocal enableextensions enabledelayedexpansion
set /p str=|| goto:eof
(set LF=^
)
for /f "delims==" %%a in ('set h. 2^>nul') do set "%%a="
set hexlist=
for %%L in ("!LF!") DO set tempstr=!str:%%=%%~L!
for /f "skip=1 tokens=* delims=" %%a in ("!tempstr!") do (
set "cur=%%a"
set hex=!cur:~0,2!
if not defined h.!hexlist! (
set h.!hexlist!=1
set hexlist=!hexlist! !hex!
)
)
set hex=%temp%\hex.txt
echo/!hexlist!>"%hex%"
certutil -f -decodehex "%hex%" "%hex%" >nul
for /f "tokens=2 delims=:." %%G in ('chcp') do set _codepage=%%G
chcp 437>nul
set/p tempstr=<"%hex%"&& (
set cur=0
for %%i in (!hexlist!) do (
for %%j in ("!cur!") do for %%p in ("%%%%i=!tempstr:~%%~j,1!") do set str=!str:%%~p!
set /a cur+=1
)
) || for %%i in (!hexlist!) do set str=!str:%%%%i=!
:print
echo/!str!>"%hex%"
chcp 65001>nul
type "%hex%"
del "%hex%"
chcp %_codepage%>nul
ANSI
This solution only support characters of codes 32-126. The rest of the code is same as above.
...
certutil -f -decodehex "%hex%" "%hex%" >nul
del "%hex%"
for %%i in (!hexlist!) do (
set /a dec=0x%%i
cmd/cexit !dec!
for %%p in ("%%%%i=!=exitcodeascii!") do set str=!str:%%~p!
)
echo !str!
' ============================================
' https://rosettacode.org/wiki/URL_decoding
' BazzBasic: https://github.com/EkBass/BazzBasic
' ============================================
' Convert two-char hex string to decimal value
DEF FN HexVal$(h$)
LET digits$ = "0123456789ABCDEF"
LET hi$ = INSTR(digits$, UCASE(LEFT(h$, 1))) - 1
LET lo$ = INSTR(digits$, UCASE(RIGHT(h$, 1))) - 1
RETURN hi$ * 16 + lo$
END DEF
' Decode a percent-encoded URL string
DEF FN UrlDecode$(s$)
LET result$ = ""
LET i$ = 1
WHILE i$ <= LEN(s$)
IF MID(s$, i$, 1) = "%" THEN
result$ = result$ + CHR(FN HexVal$(MID(s$, i$ + 1, 2)))
i$+= 3
ELSE
result$ = result$ + MID(s$, i$, 1)
i$+= 1
END IF
WEND
RETURN result$
END DEF
[inits]
LET decoded$
[main]
decoded$ = FN UrlDecode$("http%3A%2F%2Ffoo%20bar%2F")
PRINT decoded$
decoded$ = FN UrlDecode$("google.com/search?q=%60Abdu%27l-Bah%C3%A1")
PRINT decoded$
END
' Output:
' http://foo bar/
' google.com/search?q=`Abdu'l-Bahá
' Note: %C3%A1 encodes á as a UTF-8 two-byte sequence.
' CHR() is single-byte, so multi-byte UTF-8 chars render as two raw bytes.
PRINT FNurldecode("http%3A%2F%2Ffoo%20bar%2F")
END
DEF FNurldecode(url$)
LOCAL i%
REPEAT
i% = INSTR(url$, "%", i%+1)
IF i% THEN
url$ = LEFT$(url$,i%-1) + \
\ CHR$EVAL("&"+FNupper(MID$(url$,i%+1,2))) + \
\ MID$(url$,i%+3)
ENDIF
UNTIL i% = 0
= url$
DEF FNupper(A$)
LOCAL A%,C%
FOR A% = 1 TO LEN(A$)
C% = ASCMID$(A$,A%)
IF C% >= 97 IF C% <= 122 MID$(A$,A%,1) = CHR$(C%-32)
NEXT
= A$
- Output:
http://foo bar/
( ( decode
= decoded hexcode notencoded
. :?decoded
& whl
' ( @(!arg:?notencoded "%" (% %:?hexcode) ?arg)
& !decoded !notencoded chr$(x2d$!hexcode):?decoded
)
& str$(!decoded !arg)
)
& out$(decode$http%3A%2F%2Ffoo%20bar%2F)
);- Output:
http://foo bar/
#include <stdio.h>
#include <string.h>
inline int ishex(int x)
{
return (x >= '0' && x <= '9') ||
(x >= 'a' && x <= 'f') ||
(x >= 'A' && x <= 'F');
}
int decode(const char *s, char *dec)
{
char *o;
const char *end = s + strlen(s);
int c;
for (o = dec; s <= end; o++) {
c = *s++;
if (c == '+') c = ' ';
else if (c == '%' && ( !ishex(*s++) ||
!ishex(*s++) ||
!sscanf(s - 2, "%2x", &c)))
return -1;
if (dec) *o = c;
}
return o - dec;
}
int main()
{
const char *url = "http%3A%2F%2ffoo+bar%2fabcd";
char out[strlen(url) + 1];
printf("length: %d\n", decode(url, 0));
puts(decode(url, out) < 0 ? "bad string" : out);
return 0;
}
using System;
namespace URLEncode
{
internal class Program
{
private static void Main(string[] args)
{
Console.WriteLine(Decode("http%3A%2F%2Ffoo%20bar%2F"));
}
private static string Decode(string uri)
{
return Uri.UnescapeDataString(uri);
}
}
}
- Output:
http://foo bar/
#include <string>
#include "Poco/URI.h"
#include <iostream>
int main( ) {
std::string encoded( "http%3A%2F%2Ffoo%20bar%2F" ) ;
std::string decoded ;
Poco::URI::decode ( encoded , decoded ) ;
std::cout << encoded << " is decoded: " << decoded << " !" << std::endl ;
return 0 ;
}
- Output:
http%3A%2F%2Ffoo%20bar%2F is decoded: http://foo bar/ !
USER>Write $ZConvert("http%3A%2F%2Ffoo%20bar%2F", "I", "URL")
http://foo bar/
(java.net.URLDecoder/decode "http%3A%2F%2Ffoo%20bar%2F")
console.log decodeURIComponent "http%3A%2F%2Ffoo%20bar%2F?name=Foo%20Barson"
> coffee foo.coffee
http://foo bar/?name=Foo Barson
(defun decode (string &key start)
(assert (char= (char string start) #\%))
(if (>= (length string) (+ start 3))
(multiple-value-bind (code pos)
(parse-integer string :start (1+ start) :end (+ start 3) :radix 16 :junk-allowed t)
(if (= pos (+ start 3))
(values (code-char code) pos)
(values #\% (1+ start))))
(values #\% (1+ start))))
(defun url-decode (url)
(loop with start = 0
for pos = (position #\% url :start start)
collect (subseq url start pos) into chunks
when pos
collect (multiple-value-bind (decoded next) (decode url :start pos)
(setf start next)
(string decoded))
into chunks
while pos
finally (return (apply #'concatenate 'string chunks))))
(url-decode "http%3A%2F%2Ffoo%20bar%2F")
- Output:
"http://foo bar/"
require "uri"
puts URI.decode "http%3A%2F%2Ffoo%20bar%2F"
puts URI.decode "google.com/search?q=%60Abdu%27l-Bah%C3%A1"
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
import std.stdio, std.uri;
void main() {
writeln(decodeComponent("http%3A%2F%2Ffoo%20bar%2F"));
}
http://foo bar/
program URLEncoding;
{$APPTYPE CONSOLE}
uses IdURI;
begin
Writeln(TIdURI.URLDecode('http%3A%2F%2Ffoo%20bar%2F'));
end.
urldecoding is potentially ambiguous. The online decoder at https://www.urldecoder.org/ has a pull-down menu giving a list of common disambiguation options.
DuckDB's urldecode() assumes the source is UTF-8. In the following we'll present a similar decoder, and describe the changes that would be needed to make it work by decoding each occurrence of %XX separately.
UTF-8 decoder
create or replace function hex2chr(pxx) as
replace(pxx,'%','').from_hex().decode() ;
create or replace function urldecode(url) as (
WITH RECURSIVE cte AS (
-- Base case: set xx as the first sequence of %XX encodings if any
SELECT
replace(url, '+', ' ') AS decoded,
regexp_extract(replace(url, '+', ' '), '(%[0-9A-Fa-f]{2})+') AS xx
UNION ALL
-- Recursive case: replace the xx encodings with the decoded character(s)
SELECT
replace(decoded, xx, xx.hex2chr()) AS decoded,
regexp_extract(replace(decoded, xx, ' '), '(%[0-9A-Fa-f]{2})+') as xx
FROM cte
WHERE xx != ''
)
SELECT decoded FROM cte
WHERE xx = ''
);
SELECT s, url_decode(s), urldecode(s)
FROM unnest( [
'http%3A%2F%2Ffoo%20bar%2F',
'google.com/search?q=%60Abdu%27l-Bah%C3%A1',
'%25%32%35'
]) _(s);
- Output:
┌───────────────────────────────────────────┬──────────────────────────────────┬──────────────────────────────────┐ │ s │ url_decode(s) │ urldecode(s) │ │ varchar │ varchar │ varchar │ ├───────────────────────────────────────────┼──────────────────────────────────┼──────────────────────────────────┤ │ http%3A%2F%2Ffoo%20bar%2F │ http://foo bar/ │ http://foo bar/ │ │ google.com/search?q=%60Abdu%27l-Bah%C3%A1 │ google.com/search?q=`Abdu'l-Bahá │ google.com/search?q=`Abdu'l-Bahá │ │ %25%32%35 │ %25 │ %25 │ └───────────────────────────────────────────┴──────────────────────────────────┴──────────────────────────────────┘
One-at-a-time decoding
Remove the occurrences of + in the two calls to regexp_extract() above, i.e. change '(%[0-9A-Fa-f]{2})+' to '%[0-9A-Fa-f]{2}'.
- Output:
Notice that in this case, urldecode() decodes '%C3%A1' as the two characters corresponding to %C3 and %A1, i.e. `á`.
┌───────────────────────────────────────────┬──────────────────────────────────┬───────────────────────────────────┐ │ s │ url_decode(s) │ urldecode(s) │ │ varchar │ varchar │ varchar │ ├───────────────────────────────────────────┼──────────────────────────────────┼───────────────────────────────────┤ │ http%3A%2F%2Ffoo%20bar%2F │ http://foo bar/ │ http://foo bar/ │ │ google.com/search?q=%60Abdu%27l-Bah%C3%A1 │ google.com/search?q=`Abdu'l-Bahá │ google.com/search?q=`Abdu'l-Bahá │ │ %25%32%35 │ %25 │ %25 │ └───────────────────────────────────────────┴──────────────────────────────────┴───────────────────────────────────┘
func fromhex s$ .
n = number ("0x" & s$)
if error = 1 : return -1
return n
.
func$ utf8dec b[] .
ind = 1
while ind <= len b[]
n = b[ind]
if n < 0x80
cnt = 0
elif n >= 0xf0
cnt = 3
n = bitand n 0x7
elif n >= 0xe0
cnt = 2
n = bitand n 0xf
elif n >= 0xc0
cnt = 1
n = bitand n 0x1f
else
return ""
.
for i = 1 to cnt
h = b[ind + i]
if bitand h 0xc0 <> 0x80 : return ""
h = bitand h 0x3f
n = n * 64 + h
.
ind += cnt + 1
res$ &= strchar n
.
return res$
.
func$ url2decode s$ .
c$[] = strchars s$
lng = len c$[]
ind = 1
while ind < lng
if c$[ind] = "%"
b[] = [ ]
while ind <= lng and c$[ind] = "%"
if ind + 2 > lng : return ""
n = fromhex (c$[ind + 1] & c$[ind + 2])
if n = -1 : return ""
b[] &= n
ind += 3
.
res$ &= utf8dec b[]
else
res$ &= c$[ind]
ind += 1
.
.
return res$
.
print url2decode "https%3A%2F%2Fbn%2Ewikipedia%2Eorg%2Fwiki%2F%E0%A6%B0%E0%A7%8B%E0%A6%B8%E0%A7%87%E0%A6%9F%E0%A6%BE%5F%E0%A6%95%E0%A7%8B%E0%A6%A1"- Output:
https://bn.wikipedia.org/wiki/রোসেটা_কোড
IO.inspect URI.decode("http%3A%2F%2Ffoo%20bar%2F")
IO.inspect URI.decode("google.com/search?q=%60Abdu%27l-Bah%C3%A1")
- Output:
"http://foo bar/" "google.com/search?q=`Abdu'l-Bahá"
Built in.
34> http_uri:decode("http%3A%2F%2Ffoo%20bar%2F").
"http://foo bar/"
open System
let decode uri = Uri.UnescapeDataString(uri)
[<EntryPoint>]
let main argv =
printfn "%s" (decode "http%3A%2F%2Ffoo%20bar%2F")
0
USING: io kernel urls.encoding ;
IN: rosetta-code.url-decoding
"http%3A%2F%2Ffoo%20bar%2F"
"google.com/search?q=%60Abdu%27l-Bah%C3%A1"
[ url-decode print ] bi@
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
function urlDecode(data: String): AnsiString;
var
ch: Char;
pos, skip: Integer;
begin
pos := 0;
skip := 0;
Result := '';
for ch in data do begin
if skip = 0 then begin
if (ch = '%') and (pos < data.length -2) then begin
skip := 2;
Result := Result + AnsiChar(Hex2Dec('$' + data[pos+2] + data[pos+3]));
end else begin
Result := Result + ch;
end;
end else begin
skip := skip - 1;
end;
pos := pos +1;
end;
end;
Const alphanum = "0123456789abcdefghijklmnopqrstuvwxyz"
Function ToDecimal (cadena As String, base_ As Uinteger) As Uinteger
Dim As Uinteger i, n, result = 0
Dim As Uinteger inlength = Len(cadena)
For i = 1 To inlength
n = Instr(alphanum, Mid(Lcase(cadena),i,1)) - 1
n *= (base_^(inlength-i))
result += n
Next
Return result
End Function
Function url2string(cadena As String) As String
Dim As String c, nc, res
For j As Integer = 1 To Len(cadena)
c = Mid(cadena, j, 1)
If c = "%" Then
nc = Chr(ToDecimal((Mid(cadena, j+1, 2)), 16))
res &= nc
j += 2
Else
res &= c
End If
Next j
Return res
End Function
Dim As String URL = "http%3A%2F%2Ffoo%20bar%2F"
Print "Supplied URL '"; URL; "'"
Print "URL decoding '"; url2string(URL); "'"
URL = "google.com/search?q=%60Abdu%27l-Bah%C3%A1"
Print !"\nSupplied URL '"; URL; "'"
Print "URL decoding '"; url2string(URL); "'"
Sleep
local fn DecodeURL( encodedStr as CFStringRef ) as CFStringRef
end fn = fn StringByRemovingPercentEncoding( encodedStr )
print fn DecodeURL( @"http%3A%2F%2Ffoo%20bar%2F" )
print fn DecodeURL( @"google.com/search?q=%60Abdu%27l-Bah%C3%A1" )
print fn DecodeURL( @"%25%32%35" )
HandleEventshttp://foo bar/ google.com/search?q=`Abdu'l-Bahá %25
While the default is to decode parameters as UTF-8 (which is the W3C recommendation,) the characters may have been encoded in another encoding scheme, and this can be handled correctly.
URLDecode["google.com/search?q=%60Abdu%27l-Bah%C3%A1","UTF8"]package main
import (
"fmt"
"log"
"net/url"
)
func main() {
for _, escaped := range []string{
"http%3A%2F%2Ffoo%20bar%2F",
"google.com/search?q=%60Abdu%27l-Bah%C3%A1",
} {
u, err := url.QueryUnescape(escaped)
if err != nil {
log.Println(err)
continue
}
fmt.Println(u)
}
}
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
assert URLDecoder.decode('http%3A%2F%2Ffoo%20bar%2F') == 'http://foo bar/'
import qualified Data.Char as Char
urlDecode :: String -> Maybe String
urlDecode [] = Just []
urlDecode ('%':xs) =
case xs of
(a:b:xss) ->
urlDecode xss
>>= return . ((Char.chr . read $ "0x" ++ [a,b]) :)
_ -> Nothing
urlDecode ('+':xs) = urlDecode xs >>= return . (' ' :)
urlDecode (x:xs) = urlDecode xs >>= return . (x :)
main :: IO ()
main = putStrLn . maybe "Bad decode" id $ urlDecode "http%3A%2F%2Ffoo%20bar%2F"
- Output:
http://foo bar/
Another approach:
import Data.Char (chr)
import Data.List.Split (splitOn)
deCode :: String -> String
deCode url =
let ps = splitOn "%" url
in concat $
head ps :
((\(a, b) -> (chr . read) (mappend "0x" a) : b) <$> (splitAt 2 <$> tail ps))
-- TEST ------------------------------------------------------------------------
main :: IO ()
main = putStrLn $ deCode "http%3A%2F%2Ffoo%20bar%2F"
- Output:
http://foo bar/
- Output:
encoded = "http%3A%2F%2Ffoo%20bar%2F" decoded = "http://foo bar/"
J does not have a native urldecode (until version 7 when the jhs ide addon includes a jurldecode).
Here is an implementation:
require'strings convert'
urldecode=: rplc&(~.,/;"_1&a."2(,:tolower)'%',.toupper hfd i.#a.)
Example use:
urldecode 'http%3A%2F%2Ffoo%20bar%2F'
http://foo bar/
Note that an earlier implementation assumed the j6 implementation of hfd which where hexadecimal letters resulting from hfd were upper case. J8, in contrast, provides a lower case result from hfd. The addition of toupper guarantees the case insensitivity required by RFC 3986 regardless of which version of J you are using. As the parenthesized expression containing hfd is only evaluated at definition time, there's no performance penalty from the use of toupper.
Example use:
urldecode 'google.com/search?q=%60Abdu%27l-Bah%C3%A1'
google.com/search?q=`Abdu'l-Bahá
Java offers the URLDecoder and URLEncoder classes for this specific task.
import java.net.URLDecoder;
import java.nio.charset.StandardCharsets;
URLDecoder.decode("http%3A%2F%2Ffoo%20bar%2F", StandardCharsets.UTF_8)
Alternately, you could use a regular expression capture
import java.util.regex.Matcher;
import java.util.regex.Pattern;
String decode(String string) {
Pattern pattern = Pattern.compile("%([A-Za-z\\d]{2})");
Matcher matcher = pattern.matcher(string);
StringBuilder decoded = new StringBuilder(string);
char character;
int start, end, offset = 0;
while (matcher.find()) {
character = (char) Integer.parseInt(matcher.group(1), 16);
/* offset the matched index since were adjusting the string */
start = matcher.start() - offset;
end = matcher.end() - offset;
decoded.replace(start, end, String.valueOf(character));
offset += 2;
}
return decoded.toString();
}
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
decodeURIComponent("http%3A%2F%2Ffoo%20bar%2F")
If your jq already has "until", then the definition given below may be omitted.
# Emit . and stop as soon as "condition" is true.
def until(condition; next):
def u: if condition then . else (next|u) end;
u;
def url_decode:
# The helper function converts the input string written in the given
# "base" to an integer
def to_i(base):
explode
| reverse
| map(if 65 <= . and . <= 90 then . + 32 else . end) # downcase
| map(if . > 96 then . - 87 else . - 48 end) # "a" ~ 97 => 10 ~ 87
| reduce .[] as $c
# base: [power, ans]
([1,0]; (.[0] * base) as $b | [$b, .[1] + (.[0] * $c)]) | .[1];
. as $in
| length as $length
| [0, ""] # i, answer
| until ( .[0] >= $length;
.[0] as $i
| if $in[$i:$i+1] == "%"
then [ $i + 3, .[1] + ([$in[$i+1:$i+3] | to_i(16)] | implode) ]
else [ $i + 1, .[1] + $in[$i:$i+1] ]
end)
| .[1]; # answerExample:
"http%3A%2F%2Ffoo%20bar%2F" | url_decode- Output:
"http://foo bar/"
using URIParser
enc = "http%3A%2F%2Ffoo%20bar%2F"
dcd = unescape(enc)
println(enc, " => ", dcd)
- Output:
http%3A%2F%2Ffoo%20bar%2F => http://foo bar/
// version 1.1.2
import java.net.URLDecoder
fun main(args: Array<String>) {
val encoded = arrayOf("http%3A%2F%2Ffoo%20bar%2F", "google.com/search?q=%60Abdu%27l-Bah%C3%A1")
for (e in encoded) println(URLDecoder.decode(e, "UTF-8"))
}
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
url_decode()
{
decode="${*//+/ }"
eval print -r -- "\$'${decode//'%'@(??)/'\'x\1"'\$'"}'" 2>/dev/null
}
url_decode "http%3A%2F%2Ffoo%20bar%2F"
url_decode "google.com/search?q=%60Abdu%27l-Bah%C3%A1"
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
Currently lambdatalk has no builtin primitive for decoding URLs. Let's define it using Javascript.
1) define a new javascript primitive:
{script
LAMBDATALK.DICT['decodeURIComponent'] = function() {
return decodeURIComponent( arguments[0].trim() );
};
}
2) and use it:
{decodeURIComponent http%3A%2F%2Ffoo%20bar%2F}
-> http://foo bar/
val finish = fn(s) {
b2s(map(
less(split(s, delim="%"), of=1),
by=fn(x) { number x, fmt=16 },
))
}
val decode = fn(s) {
replace(
s,
by=re/(%[0-9A-Fa-f]{2})+/,
with=finish,
)
}
writeln decode("http%3A%2F%2Fno%20more%20foo%20bars%20please%2F")
writeln decode("google.com/search?q=%22unbroken%20string%22")- Output:
https://no more foo bars please/ google.com/search?q="unbroken string"
bytes('http%3A%2F%2Ffoo%20bar%2F') -> decodeurl
-> http://foo bar/
dim lookUp$( 256)
for i =0 to 256
lookUp$( i) ="%" +dechex$( i)
next i
url$ ="http%3A%2F%2Ffoo%20bar%2F"
print "Supplied URL '"; url$; "'"
print "As string '"; url2string$( url$); "'"
end
function url2string$( i$)
for j =1 to len( i$)
c$ =mid$( i$, j, 1)
if c$ ="%" then
nc$ =chr$( hexdec( mid$( i$, j +1, 2)))
url2string$ =url2string$ +nc$
j =j +2
else
url2string$ =url2string$ +c$
end if
next j
end functionSupplied URL 'http%3A%2F%2Ffoo%20bar%2F' As string 'http://foo bar/'
----------------------------------------
-- URL decodes a string
-- @param {string} str
-- @return {string}
----------------------------------------
on urldecode (str)
res = ""
ba = bytearray()
len = str.length
repeat with i = 1 to len
c = str.char[i]
if (c = "%") then
-- fastest hex-to-dec conversion hack based on Lingo's rgb object
ba.writeInt8(rgb(str.char[i+1..i+2]).blue)
i = i + 2
else if (c = "+") then
ba.writeInt8(32)
else
ba.writeInt8(chartonum(c))
end if
end repeat
ba.position = 1
return ba.readRawString(ba.length)
endput urldecode("http%3A%2F%2Ffoo%20bar%2F")
put urldecode("google.com/search?q=%60Abdu%27l-Bah%C3%A1")- Output:
-- "http://foo bar/" -- "google.com/search?q=`Abdu'l-Bahá"
put urlDecode("http%3A%2F%2Ffoo%20bar%2F") & cr & \
urlDecode("google.com/search?q=%60Abdu%27l-Bah%C3%A1")Results
http://foo bar/
google.com/search?q=`Abdu'l-Bah√°
function decodeChar(hex)
return string.char(tonumber(hex,16))
end
function decodeString(str)
local output, t = string.gsub(str,"%%(%x%x)",decodeChar)
return output
end
-- will print "http://foo bar/"
print(decodeString("http%3A%2F%2Ffoo%20bar%2F"))
Function Len(string) return length in words, so a value 1.5 means 3 bytes
We can add strings with half word at the end of a series of words.
A$=str$("A") has a length of 0.5
b$=chr$(a$) revert bytes to words adding zeroes after each character
Module CheckIt {
Function decodeUrl$(a$) {
DIM a$()
a$()=Piece$(a$, "%")
if len(a$())=1 then =str$(a$):exit
k=each(a$(),2)
\\ convert to one byte per character using str$(string)
acc$=str$(a$(0))
While k {
\\ chr$() convert to UTF16LE
\\ str$() convert to ANSI using locale (can be 1033 we can set it before as Locale 1033)
\\ so chr$(0x93) give 0x201C
\\ str$(chr$(0x93)) return one byte 93 in ANSI as string of one byte length
\\ numbers are for UTF-8 so we have to preserve them
acc$+=str$(Chr$(Eval("0x"+left$(a$(k^),2)))+Mid$(a$(k^),3))
}
=acc$
}
\\ decode from utf8
final$=DecodeUrl$("google.com/search?q=%60Abdu%27l-Bah%C3%A1")
Print string$(final$ as utf8dec)="google.com/search?q=`Abdu'l-Bahá"
final$=DecodeUrl$("http%3A%2F%2Ffoo%20bar%2F")
Print string$(final$ as utf8dec)="http://foo bar/"
}
CheckItStringTools:-Decode("http%3A%2F%2Ffoo%20bar%2F", encoding=percent);
- Output:
"http://foo bar/"
URLDecoding[url_] :=
StringReplace[url, "%" ~~ x_ ~~ y_ :> FromDigits[x ~~ y, 16]] //.
StringExpression[x___, Longest[n__Integer], y___] :>
StringExpression[x, FromCharacterCode[{n}, "UTF8"], y]
Example use:
URLDecoding["http%3A%2F%2Ffoo%20bar%2F"]
- Output:
http://foo bar/
Using the built-in URLDecode (http://reference.wolfram.com/language/ref/URLDecode.html) function:
- Output:
In[]:= URLDecode["http%3A%2F%2Ffoo%20bar%2F"]
Out[]= "http://foo bar/"
In[]:= URLDecode["google.com/search?q=%60Abdu%27l-Bah%C3%A1"]
Out[]= "google.com/search?q=`Abdu'l-Bahá"
In[]:= URLDecode[{"Kurt+G%C3%B6del", "Paul+Erd%C5%91s"}]
Out[]= {"Kurt Gödel", "Paul Erdős"}
function u = urldecoding(s)
u = '';
k = 1;
while k<=length(s)
if s(k) == '%' && k+2 <= length(s)
u = sprintf('%s%c', u, char(hex2dec(s((k+1):(k+2)))));
k = k + 3;
else
u = sprintf('%s%c', u, s(k));
k = k + 1;
end
end
end
Usage:
octave:3> urldecoding('http%3A%2F%2Ffoo%20bar%2F')
ans = http://foo bar/
import "fmt"
import "utf8"
urlDecode = function(enc)
bytes = []
i = 0
while i < enc.len
c = enc[i]
if c == "%" then
b = fmt.atoi(enc[i+1:i+3], 16)
bytes.push b
i += 3
else
bytes.push code(c)
i += 1
end if
end while
return utf8.decode(bytes)
end function
encs = [
"http%3A%2F%2Ffoo%20bar%2F",
"google.com/search?q=%60Abdu%27l-Bah%C3%A1",
]
for enc in encs
print urlDecode(enc)
end for
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
/* NetRexx */
options replace format comments java crossref savelog symbols nobinary
url = [ -
'http%3A%2F%2Ffoo%20bar%2F', -
'mailto%3A%22Ivan%20Aim%22%20%3Civan%2Eaim%40email%2Ecom%3E', -
'%6D%61%69%6C%74%6F%3A%22%49%72%6D%61%20%55%73%65%72%22%20%3C%69%72%6D%61%2E%75%73%65%72%40%6D%61%69%6C%2E%63%6F%6D%3E' -
]
loop u_ = 0 to url.length - 1
say url[u_]
say DecodeURL(url[u_])
say
end u_
return
method DecodeURL(arg) public static
Parse arg encoded
decoded = ''
PCT = '%'
loop label e_ while encoded.length() > 0
parse encoded head (PCT) +1 code +2 tail
decoded = decoded || head
select
when code.strip('T').length() = 2 & code.datatype('X') then do
code = code.x2c()
decoded = decoded || code
end
when code.strip('T').length() \= 0 then do
decoded = decoded || PCT
tail = code || tail
end
otherwise do
nop
end
end
encoded = tail
end e_
return decoded
- Output:
http%3A%2F%2Ffoo%20bar%2F http://foo bar/ mailto%3A%22Ivan%20Aim%22%20%3Civan%2Eaim%40email%2Ecom%3E mailto:"Ivan Aim" <ivan.aim@email.com> %6D%61%69%6C%74%6F%3A%22%49%72%6D%61%20%55%73%65%72%22%20%3C%69%72%6D%61%2E%75%73%65%72%40%6D%61%69%6C%2E%63%6F%6D%3E mailto:"Irma User" <irma.user@mail.com>
;; universal decoder, works for ASCII and UTF-8
;; (source http://www.newlisp.org/index.cgi?page=Code_Snippets)
(define (url-decode url (opt nil))
(if opt (replace "+" url " "))
(replace "%([0-9a-f][0-9a-f])" url (pack "b" (int $1 0 16)) 1))
(url-decode "http%3A%2F%2Ffoo%20bar%2F")
import cgi
echo decodeUrl("http%3A%2F%2Ffoo%20bar%2F")
- Output:
http://foo bar/
MODULE URLDecoding;
IMPORT
URI := URI:String,
Out := NPCT:Console;
BEGIN
Out.String(URI.Unescape("http%3A%2F%2Ffoo%20bar%2F"));Out.Ln;
Out.String(URI.Unescape("google.com/search?q=%60Abdu%27l-Bah%C3%A1"));Out.Ln;
END URLDecoding.
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
class UrlDecode {
function : Main(args : String[]) ~ Nil {
Net.UrlUtility->Decode("http%3A%2F%2Ffoo%20bar%2F")->PrintLine();
}
}NSString *encoded = @"http%3A%2F%2Ffoo%20bar%2F";
NSString *normal = [encoded stringByReplacingPercentEscapesUsingEncoding:NSUTF8StringEncoding];
NSLog(@"%@", normal);
NSString *encoded = @"http%3A%2F%2Ffoo%20bar%2F";
NSString *normal = [encoded stringByRemovingPercentEncoding];
NSLog(@"%@", normal);
Using the library ocamlnet from the interactive loop:
$ ocaml
# #use "topfind";;
# #require "netstring";;
# Netencoding.Url.decode "http%3A%2F%2Ffoo%20bar%2F" ;;
- : string = "http://foo bar/"
While the implementation shown for Rexx will also work with ooRexx, this version uses ooRexx syntax to invoke the built-in functions.
/* Rexx */
X = 0
url. = ''
X = X + 1; url.0 = X; url.X = 'http%3A%2F%2Ffoo%20bar%2F'
X = X + 1; url.0 = X; url.X = 'mailto%3A%22Ivan%20Aim%22%20%3Civan%2Eaim%40email%2Ecom%3E'
X = X + 1; url.0 = X; url.X = '%6D%61%69%6C%74%6F%3A%22%49%72%6D%61%20%55%73%65%72%22%20%3C%69%72%6D%61%2E%75%73%65%72%40%6D%61%69%6C%2E%63%6F%6D%3E'
Do u_ = 1 to url.0
Say url.u_
Say DecodeURL(url.u_)
Say
End u_
Exit
DecodeURL: Procedure
Parse Arg encoded
decoded = ''
PCT = '%'
Do label e_ while encoded~length() > 0
Parse Var encoded head (PCT) +1 code +2 tail
decoded = decoded || head
Select
when code~strip('T')~length() = 2 & code~datatype('X') then Do
code = code~x2c()
decoded = decoded || code
End
when code~strip('T')~length() \= 0 then Do
decoded = decoded || PCT
tail = code || tail
End
otherwise
Nop
End
encoded = tail
End e_
Return decoded
- Output:
http%3A%2F%2Ffoo%20bar%2F http://foo bar/ mailto%3A%22Ivan%20Aim%22%20%3Civan%2Eaim%40email%2Ecom%3E mailto:"Ivan Aim" <ivan.aim@email.com> %6D%61%69%6C%74%6F%3A%22%49%72%6D%61%20%55%73%65%72%22%20%3C%69%72%6D%61%2E%75%73%65%72%40%6D%61%69%6C%2E%63%6F%6D%3E mailto:"Irma User" <irma.user@mail.com>
uses System;
function URLDecode(s: string) := Uri.UnescapeDataString(s);
begin
Println(URLDecode('http%3A%2F%2Ffoo%20bar%2F'));
Println(URLDecode('google.com/search?q=%60Abdu%27l-Bah%C3%A1'));
Println(URLDecode('%25%32%35'));
end.
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá %25
sub urldecode {
my $s = shift;
$s =~ tr/\+/ /;
$s =~ s/\%([A-Fa-f0-9]{2})/pack('C', hex($1))/eg;
return $s;
}
print urldecode('http%3A%2F%2Ffoo+bar%2F')."\n";
#!/usr/bin/perl -w
use strict ;
use URI::Escape ;
my $encoded = "http%3A%2F%2Ffoo%20bar%2F" ;
my $unencoded = uri_unescape( $encoded ) ;
print "The unencoded string is $unencoded !\n" ;
-- demo\rosetta\decode_url.exw
with javascript_semantics
function decode_url(string s)
integer skip = 0
string res = ""
for i=1 to length(s) do
if skip then
skip -= 1
else
integer ch = s[i]
if ch='%' then
sequence scanres = {}
if i+2<=length(s) then
scanres = scanf("#"&s[i+1..i+2],"%x")
end if
if length(scanres)!=1 then
return "decode error"
end if
skip = 2
ch = scanres[1][1]
elsif ch='+' then
ch = ' '
end if
res &= ch
end if
end for
return res
end function
printf(1,"%s\n",{decode_url("http%3A%2F%2Ffoo%20bar%2F")})
printf(1,"%s\n",{decode_url("google.com/search?q=%60Abdu%27l-Bah%C3%A1")})
wait_key()
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
<?php
$encoded = "http%3A%2F%2Ffoo%20bar%2F";
$unencoded = rawurldecode($encoded);
echo "The unencoded string is $unencoded !\n";
?>
: (ht:Pack (chop "http%3A%2F%2Ffoo%20bar%2F") T) -> "http://foo bar/"
void main()
{
array encoded_urls = ({
"http%3A%2F%2Ffoo%20bar%2F",
"google.com/search?q=%60Abdu%27l-Bah%C3%A1"
});
foreach(encoded_urls, string url) {
string decoded = Protocols.HTTP.uri_decode( url );
write( string_to_utf8(decoded) +"\n" ); // Assume sink does UTF8
}
}
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
[System.Web.HttpUtility]::UrlDecode("http%3A%2F%2Ffoo%20bar%2F")
- Output:
http://foo bar/
URL$ = URLDecoder("http%3A%2F%2Ffoo%20bar%2F")
Debug URL$ ; http://foo bar/
#Python 2.X
import urllib
print urllib.unquote("http%3A%2F%2Ffoo%20bar%2F")
#Python 3.5+
from urllib.parse import unquote
print(unquote('http%3A%2F%2Ffoo%20bar%2F'))
URLdecode("http%3A%2F%2Ffoo%20bar%2F")
#lang racket
(require net/uri-codec)
(uri-decode "http%3A%2F%2Ffoo%20bar%2F")
(formerly Perl 6)
my @urls = < http%3A%2F%2Ffoo%20bar%2F
google.com/search?q=%60Abdu%27l-Bah%C3%A1 >;
say .subst( :g,
/ [ '%' ( <xdigit> ** 2 ) ]+ / ,
{ Blob.new((:16(~$_) for $0)).decode }
) for @urls;
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
Rebol [
title: "Rosetta code: URL decoding"
file: %URL_decoding.r3
url: https://rosettacode.org/wiki/URL_decoding
]
foreach [src expected] [
"http%3A%2F%2Ffoo%20bar%2F"
"http://foo bar/"
"google.com/search?q=%60Abdu%27l-Bah%C3%A1"
"google.com/search?q=`Abdu'l-Bahá"
"%25%32%35"
"%25"
][
probe src
probe url: dehex src
print either expected = url ["OK"]["FAILED!"]
print ""
]
- Output:
"http%3A%2F%2Ffoo%20bar%2F" "http://foo bar/" OK "google.com/search?q=%60Abdu%27l-Bah%C3%A1" "google.com/search?q=`Abdu'l-Bahá" OK "%25%32%35" "%25" OK
>> dehex "http%3A%2F%2Ffoo%20bar%2F"
== "http://foo bar/"
>> dehex "google.com/search?q=%60Abdu%27l-Bah%C3%A1"
== "google.com/search?q=`Abdu'l-Bahá"
This is provided by the casket library (used for web app development).
create buffer 32000 allot
{{
create bit 5 allot
: extract ( $c-$a ) drop @+ bit ! @+ bit 1+ ! bit ;
: render ( $c-$n )
dup '+ = [ drop 32 ] ifTrue
dup 13 = [ drop 32 ] ifTrue
dup 10 = [ drop 32 ] ifTrue
dup '% = [ extract hex toNumber decimal ] ifTrue ;
: <decode> ( $-$ ) repeat @+ 0; render ^buffer'add again ;
---reveal---
: decode ( $- ) buffer ^buffer'set <decode> drop ;
}}
"http%3A%2F%2Ffoo%20bar%2F" decode buffer putsversion 1
Tested with the ooRexx and Regina interpreters.
/* Rexx */
Do
X = 0
url. = ''
X = X + 1; url.0 = X; url.X = 'http%3A%2F%2Ffoo%20bar%2F'
X = X + 1; url.0 = X; url.X = 'mailto%3A%22Ivan%20Aim%22%20%3Civan%2Eaim%40email%2Ecom%3E'
X = X + 1; url.0 = X; url.X = '%6D%61%69%6C%74%6F%3A%22%49%72%6D%61%20%55%73%65%72%22%20%3C%69%72%6D%61%2E%75%73%65%72%40%6D%61%69%6C%2E%63%6F%6D%3E'
Do u_ = 1 to url.0
Say url.u_
Say DecodeURL(url.u_)
Say
End u_
Return
End
Exit
DecodeURL:
Procedure
Do
Parse Arg encoded
decoded = ''
PCT = '%'
Do while length(encoded) > 0
Parse Var encoded head (PCT) +1 code +2 tail
decoded = decoded || head
Select
When length(strip(code, 'T')) = 2 & datatype(code, 'X') then Do
code = x2c(code)
decoded = decoded || code
End
When length(strip(code, 'T')) \= 0 then Do
decoded = decoded || PCT
tail = code || tail
End
Otherwise Do
Nop
End
End
encoded = tail
End
Return decoded
End
Exit
- Output:
http%3A%2F%2Ffoo%20bar%2F http://foo bar/ mailto%3A%22Ivan%20Aim%22%20%3Civan%2Eaim%40email%2Ecom%3E mailto:"Ivan Aim" <ivan.aim@email.com> %6D%61%69%6C%74%6F%3A%22%49%72%6D%61%20%55%73%65%72%22%20%3C%69%72%6D%61%2E%75%73%65%72%40%6D%61%69%6C%2E%63%6F%6D%3E mailto:"Irma User" <irma.user@mail.com>
version 2
This REXX version is identical to version 1, but with superfluous and dead code removed.
/*REXX program converts a URL─encoded string ──► its original unencoded form. */
url.1='http%3A%2F%2Ffoo%20bar%2F'
url.2='mailto%3A%22Ivan%20Aim%22%20%3Civan%2Eaim%40email%2Ecom%3E'
url.3='%6D%61%69%6C%74%6F%3A%22%49%72%6D%61%20%55%73%65%72%22%20%3C%69%72%6D%61%2E%75%73%65%72%40%6D%61%69%6C%2E%63%6F%6D%3E'
URLs =3
do j=1 for URLs
say url.j
say decodeURL(url.j)
say
end /*j*/
exit
/*──────────────────────────────────────────────────────────────────────────────────────*/
decodeURL: procedure; parse arg encoded; decoded= ''
do while encoded\==''
parse var encoded head '%' +1 code +2 tail
decoded= decoded || head
L= length( strip( code, 'T') )
select
when L==2 & datatype(code, "X") then decoded= decoded || x2c(code)
when L\==0 then do; decoded= decoded'%'
tail= code || tail
end
otherwise nop
end /*select*/
encoded= tail
end /*while*/
return decoded
- output is identical to the 1st REXX version.
version 3
This REXX version is a shorter version of version 2.
/*REXX program converts & displays a URL─encoded string ──► its original unencoded form.*/
url. =
url.1='http%3A%2F%2Ffoo%20bar%2F'
url.2='mailto%3A%22Ivan%20Aim%22%20%3Civan%2Eaim%40email%2Ecom%3E'
url.3='%6D%61%69%6C%74%6F%3A%22%49%72%6D%61%20%55%73%65%72%22%20%3C%69%72%6D%61%2E%75%73%65%72%40%6D%61%69%6C%2E%63%6F%6D%3E'
do j=1 until url.j==''; say /*process each URL; display blank line.*/
say url.j /*display the original URL. */
say URLdecode(url.j) /* " " decoded " */
end /*j*/
exit /*stick a fork in it, we're all done. */
/*──────────────────────────────────────────────────────────────────────────────────────*/
URLdecode: procedure; parse arg yyy /*get encoded URL from argument list. */
yyy= translate(yyy, , '+') /*a special case for an encoded blank. */
URL=
do until yyy==''
parse var yyy plain '%' +1 code +2 yyy
URL= URL || plain
if datatype(code, 'X') then URL= URL || x2c(code)
else URL= URL'%'code
end /*until*/
return URL
- output is identical to the 1st REXX version.
Use any one of CGI.unescape or URI.decode_www_form_component. These methods also convert "+" to " ".
require 'cgi'
puts CGI.unescape("http%3A%2F%2Ffoo%20bar%2F")
# => "http://foo bar/"
require 'uri'
puts URI.decode_www_form_component("http%3A%2F%2Ffoo%20bar%2F")
# => "http://foo bar/"
URI.unescape (alias URI.unencode) still works. URI.unescape is obsolete since Ruby 1.9.2 because of problems with its sibling URI.escape.
const INPUT1: &str = "http%3A%2F%2Ffoo%20bar%2F";
const INPUT2: &str = "google.com/search?q=%60Abdu%27l-Bah%C3%A1";
fn append_frag(text: &mut String, frag: &mut String) {
if !frag.is_empty() {
let encoded = frag.chars()
.collect::<Vec<char>>()
.chunks(2)
.map(|ch| {
u8::from_str_radix(&ch.iter().collect::<String>(), 16).unwrap()
}).collect::<Vec<u8>>();
text.push_str(&std::str::from_utf8(&encoded).unwrap());
frag.clear();
}
}
fn decode(text: &str) -> String {
let mut output = String::new();
let mut encoded_ch = String::new();
let mut iter = text.chars();
while let Some(ch) = iter.next() {
if ch == '%' {
encoded_ch.push_str(&format!("{}{}", iter.next().unwrap(), iter.next().unwrap()));
} else {
append_frag(&mut output, &mut encoded_ch);
output.push(ch);
}
}
append_frag(&mut output, &mut encoded_ch);
output
}
fn main() {
println!("{}", decode(INPUT1));
println!("{}", decode(INPUT2));
}
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
import java.net.{URLDecoder, URLEncoder}
import scala.compat.Platform.currentTime
object UrlCoded extends App {
val original = """http://foo bar/"""
val encoded: String = URLEncoder.encode(original, "UTF-8")
assert(encoded == "http%3A%2F%2Ffoo+bar%2F", s"Original: $original not properly encoded: $encoded")
val percentEncoding = encoded.replace("+", "%20")
assert(percentEncoding == "http%3A%2F%2Ffoo%20bar%2F", s"Original: $original not properly percent-encoded: $percentEncoding")
assert(URLDecoder.decode(encoded, "UTF-8") == URLDecoder.decode(percentEncoding, "UTF-8"))
println(s"Successfully completed without errors. [total ${currentTime - executionStart} ms]")
}
The library encoding.s7i defines functions to handle URL respectively percent encoding. The function fromPercentEncoded decodes a percend-encoded string. The function fromUrlEncoded works like fromPercentEncoded and additionally decodes '+' with a space. Both functions return byte sequences. To decode Unicode characters it is necessary to convert them from UTF-8 with fromUtf8 afterwards.
$ include "seed7_05.s7i";
include "encoding.s7i";
const proc: main is func
begin
writeln(fromPercentEncoded("http%3A%2F%2Ffoo%20bar%2F"));
writeln(fromUrlEncoded("http%3A%2F%2Ffoo+bar%2F"));
end func;- Output:
http://foo bar/ http://foo bar/
func urldecode(str) {
str.gsub!('+', ' ');
str.gsub!(/\%([A-Fa-f0-9]{2})/, {|a| 'C'.pack(a.hex)});
return str;
}
say urldecode('http%3A%2F%2Ffoo+bar%2F'); # => "http://foo bar/"
import Foundation
let encoded = "http%3A%2F%2Ffoo%20bar%2F"
if let normal = encoded.stringByReplacingPercentEscapesUsingEncoding(NSUTF8StringEncoding) {
println(normal)
}
This code is careful to ensure that any untoward metacharacters in the input string still do not cause any problems.
proc urlDecode {str} {
set specialMap {"[" "%5B" "]" "%5D"}
set seqRE {%([0-9a-fA-F]{2})}
set replacement {[format "%c" [scan "\1" "%2x"]]}
set modStr [regsub -all $seqRE [string map $specialMap $str] $replacement]
return [encoding convertfrom utf-8 [subst -nobackslash -novariable $modStr]]
}
Demonstrating:
puts [urlDecode "http%3A%2F%2Ffoo%20bar%2F"]
- Output:
http://foo bar/
$$ MODE TUSCRIPT
url_encoded="http%3A%2F%2Ffoo%20bar%2F"
BUILD S_TABLE hex=":%><:><2<>2<%:"
hex=STRINGS (url_encoded,hex), hex=SPLIT(hex)
hex=DECODE (hex,hex)
url_decoded=SUBSTITUTE(url_encoded,":%><2<>2<%:",0,0,hex)
PRINT "encoded: ", url_encoded
PRINT "decoded: ", url_decoded- Output:
encoded: http%3A%2F%2Ffoo%20bar%2F decoded: http://foo bar/
urldecode() { local u="${1//+/ }"; printf '%b' "${u//%/\\x}"; }
Alternative: Replace printf '%b' "${u//%/\\x}" with echo -e "${u//%/\\x}"
Example:
urldecode http%3A%2F%2Ffoo%20bar%2F
http://foo bar/
urldecode google.com/search?q=%60Abdu%27l-Bah%C3%A1
google.com/search?q=`Abdu'l-Bahároot@
function urldecode
{
typeset encoded=$1 decoded= rest= c= c1= c2=
typeset rest2= bug='rest2=${rest}'
if [[ -z ${BASH_VERSION:-} ]]; then
typeset -i16 hex=0; typeset -i8 oct=0
# bug /usr/bin/sh HP-UX 11.00
typeset _encoded='xyz%26xyz'
rest="${_encoded#?}"
c="${_encoded%%${rest}}"
if (( ${#c} != 1 )); then
typeset qm='????????????????????????????????????????????????????????????????????????'
typeset bug='(( ${#rest} > 0 )) && typeset -L${#rest} rest2="${qm}" || rest2=${rest}'
fi
fi
rest="${encoded#?}"
eval ${bug}
c="${encoded%%${rest2}}"
encoded="${rest}"
while [[ -n ${c} ]]; do
if [[ ${c} = '%' ]]; then
rest="${encoded#?}"
eval ${bug}
c1="${encoded%%${rest2}}"
encoded="${rest}"
rest="${encoded#?}"
eval ${bug}
c2="${encoded%%${rest2}}"
encoded="${rest}"
if [[ -z ${c1} || -z ${c2} ]]; then
c="%${c1}${c2}"
echo "WARNING: invalid % encoding: ${c}" >&2
elif [[ -n ${BASH_VERSION:-} ]]; then
c="\\x${c1}${c2}"
c=$(\echo -e "${c}")
else
hex="16#${c1}${c2}"; oct=hex
c="\\0${oct#8\#}"
c=$(print -- "${c}")
fi
elif [[ ${c} = '+' ]]; then
c=' '
fi
decoded="${decoded}${c}"
rest="${encoded#?}"
eval ${bug}
c="${encoded%%${rest2}}"
encoded="${rest}"
done
if [[ -n ${BASH_VERSION:-} ]]; then
\echo -E "${decoded}"
else
print -r -- "${decoded}"
fi
}
Function RegExTest(str,patrn)
Dim regEx
Set regEx = New RegExp
regEx.IgnoreCase = True
regEx.Pattern = patrn
RegExTest = regEx.Test(str)
End Function
Function URLDecode(sStr)
Dim str,code,a0
str=""
code=sStr
code=Replace(code,"+"," ")
While len(code)>0
If InStr(code,"%")>0 Then
str = str & Mid(code,1,InStr(code,"%")-1)
code = Mid(code,InStr(code,"%"))
a0 = UCase(Mid(code,2,1))
If a0="U" And RegExTest(code,"^%u[0-9A-F]{4}") Then
str = str & ChrW((Int("&H" & Mid(code,3,4))))
code = Mid(code,7)
ElseIf a0="E" And RegExTest(code,"^(%[0-9A-F]{2}){3}") Then
str = str & ChrW((Int("&H" & Mid(code,2,2)) And 15) * 4096 + (Int("&H" & Mid(code,5,2)) And 63) * 64 + (Int("&H" & Mid(code,8,2)) And 63))
code = Mid(code,10)
ElseIf a0>="C" And a0<="D" And RegExTest(code,"^(%[0-9A-F]{2}){2}") Then
str = str & ChrW((Int("&H" & Mid(code,2,2)) And 3) * 64 + (Int("&H" & Mid(code,5,2)) And 63))
code = Mid(code,7)
ElseIf (a0<="B" Or a0="F") And RegExTest(code,"^%[0-9A-F]{2}") Then
str = str & Chr(Int("&H" & Mid(code,2,2)))
code = Mid(code,4)
Else
str = str & "%"
code = Mid(code,2)
End If
Else
str = str & code
code = ""
End If
Wend
URLDecode = str
End Function
url = "http%3A%2F%2Ffoo%20bar%C3%A8%2F"
WScript.Echo "Encoded URL: " & url & vbCrLf &_
"Decoded URL: " & UrlDecode(url)
- Output:
Encoded URL: http%3A%2F%2Ffoo%20bar%C3%A8%2F Decoded URL: http://foo barè/
import net.urllib
fn main() {
for escaped in [
"http%3A%2F%2Ffoo%20bar%2F",
"google.com/search?q=%60Abdu%27l-Bah%C3%A1",
] {
u := urllib.query_unescape(escaped)!
println(u)
}
}
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
import "./fmt" for Conv
var urlDecode = Fn.new { |enc|
var res = ""
var i = 0
while (i < enc.count) {
var c = enc[i]
if (c == "\%") {
var b = Conv.atoi(enc[i+1..i+2], 16)
res = res + String.fromByte(b)
i = i + 3
} else {
res = res + c
i = i + 1
}
}
return res
}
// We need to escape % characters in Wren as % is otherwise used for string interpolation.
var encs = [
"http\%3A\%2F\%2Ffoo\%20bar\%2F",
"google.com/search?q=\%60Abdu\%27l-Bah\%C3\%A1"
]
for (enc in encs)System.print(urlDecode.call(enc))
- Output:
http://foo bar/ google.com/search?q=`Abdu'l-Bahá
code Text=12;
string 0; \use zero-terminated strings
func Decode(S0); \Decode URL string and return its address
char S0;
char S1(80); \BEWARE: very temporary string space returned
int C, N, I, J;
[I:= 0; J:= 0;
repeat C:= S0(I); I:= I+1; \get char
if C=^% then \convert hex to char
[C:= S0(I); I:= I+1;
if C>=^a then C:= C & ~$20; \convert to uppercase
N:= C - (if C<=^9 then ^0 else ^A-10);
C:= S0(I); I:= I+1;
if C>=^a then C:= C & ~$20;
C:= N*16 + C - (if C<=^9 then ^0 else ^A-10);
];
S1(J):= C; J:= J+1; \put char in output string
until C=0;
return S1;
];
Text(0, Decode("http%3A%2F%2Ffoo%20bar%2f"))- Output:
http://foo bar/
sub decode_url$(s$)
local res$, ch$
while(s$ <> "")
ch$ = left$(s$, 1)
if ch$ = "%" then
ch$ = chr$(dec(mid$(s$, 2, 2)))
s$ = right$(s$, len(s$) - 3)
else
if ch$ = "+" ch$ = " "
s$ = right$(s$, len(s$) - 1)
endif
res$ = res$ + ch$
wend
return res$
end sub
print decode_url$("http%3A%2F%2Ffoo%20bar%2F")
print decode_url$("google.com/search?q=%60Abdu%27l-Bah%C3%A1")!ys-0
strs =::
- "http%3A%2F%2Ffoo%20bar%2F"
- "google.com/search?q=%60Abdu%27l-Bah%C3%A1"
- "%25%32%35"
defn main():
each s strs:
say: s:url-decode
defn- url-decode(s):
s2 =: s.replace(/\+/ ' ')
pat =: /((?:[^%]|%(?![0-9A-Fa-f]{2}))*)((?:%[0-9A-Fa-f]{2})*)/
parts =: re-seq(pat s2)
reduce _ "" parts:
fn(acc m):
lit =: m.1
pct =: m.2
if pct:empty?:
then: acc + lit
else: acc + lit + decode-bytes(pct)
defn- decode-bytes(pct):
hexes =: re-seq(/%([0-9A-Fa-f]{2})/ pct)
bs =: byte-array(hexes.map(\(Integer/parseInt(_.1 16))))
new: String bs "UTF-8"
- Output:
$ ys url-decoding.ys http://foo bar/ google.com/search?q=`Abdu'l-Bahá %25
"http%3A%2F%2Ffoo%20bar%2F".pump(String, // push each char through these fcns:
fcn(c){ if(c=="%") return(Void.Read,2); return(Void.Skip,c) },// %-->read 2 chars else pass through
fcn(_,b,c){ (b+c).toInt(16).toChar() }) // "%" (ignored) "3"+"1"-->0x31-->"1"- Output:
http://foo bar/
or use libCurl:
var Curl=Import.lib("zklCurl");
Curl.urlDecode("http%3A%2F%2Ffoo%20bar%2F");- Output:
http://foo bar/