bgneal@1
|
1 var LATIN_MAP = {
|
bgneal@1
|
2 'À': 'A', 'Á': 'A', 'Â': 'A', 'Ã': 'A', 'Ä': 'A', 'Å': 'A', 'Æ': 'AE', 'Ç':
|
bgneal@1
|
3 'C', 'È': 'E', 'É': 'E', 'Ê': 'E', 'Ë': 'E', 'Ì': 'I', 'Í': 'I', 'Î': 'I',
|
bgneal@1
|
4 'Ï': 'I', 'Ð': 'D', 'Ñ': 'N', 'Ò': 'O', 'Ó': 'O', 'Ô': 'O', 'Õ': 'O', 'Ö':
|
bgneal@1
|
5 'O', 'Ő': 'O', 'Ø': 'O', 'Ù': 'U', 'Ú': 'U', 'Û': 'U', 'Ü': 'U', 'Ű': 'U',
|
bgneal@1
|
6 'Ý': 'Y', 'Þ': 'TH', 'ß': 'ss', 'à':'a', 'á':'a', 'â': 'a', 'ã': 'a', 'ä':
|
bgneal@1
|
7 'a', 'å': 'a', 'æ': 'ae', 'ç': 'c', 'è': 'e', 'é': 'e', 'ê': 'e', 'ë': 'e',
|
bgneal@1
|
8 'ì': 'i', 'í': 'i', 'î': 'i', 'ï': 'i', 'ð': 'd', 'ñ': 'n', 'ò': 'o', 'ó':
|
bgneal@1
|
9 'o', 'ô': 'o', 'õ': 'o', 'ö': 'o', 'ő': 'o', 'ø': 'o', 'ù': 'u', 'ú': 'u',
|
bgneal@1
|
10 'û': 'u', 'ü': 'u', 'ű': 'u', 'ý': 'y', 'þ': 'th', 'ÿ': 'y'
|
bgneal@1
|
11 }
|
bgneal@1
|
12 var LATIN_SYMBOLS_MAP = {
|
bgneal@1
|
13 '©':'(c)'
|
bgneal@1
|
14 }
|
bgneal@1
|
15 var GREEK_MAP = {
|
bgneal@1
|
16 'α':'a', 'β':'b', 'γ':'g', 'δ':'d', 'ε':'e', 'ζ':'z', 'η':'h', 'θ':'8',
|
bgneal@1
|
17 'ι':'i', 'κ':'k', 'λ':'l', 'μ':'m', 'ν':'n', 'ξ':'3', 'ο':'o', 'π':'p',
|
bgneal@1
|
18 'ρ':'r', 'σ':'s', 'τ':'t', 'υ':'y', 'φ':'f', 'χ':'x', 'ψ':'ps', 'ω':'w',
|
bgneal@1
|
19 'ά':'a', 'έ':'e', 'ί':'i', 'ό':'o', 'ύ':'y', 'ή':'h', 'ώ':'w', 'ς':'s',
|
bgneal@1
|
20 'ϊ':'i', 'ΰ':'y', 'ϋ':'y', 'ΐ':'i',
|
bgneal@1
|
21 'Α':'A', 'Β':'B', 'Γ':'G', 'Δ':'D', 'Ε':'E', 'Ζ':'Z', 'Η':'H', 'Θ':'8',
|
bgneal@1
|
22 'Ι':'I', 'Κ':'K', 'Λ':'L', 'Μ':'M', 'Ν':'N', 'Ξ':'3', 'Ο':'O', 'Π':'P',
|
bgneal@1
|
23 'Ρ':'R', 'Σ':'S', 'Τ':'T', 'Υ':'Y', 'Φ':'F', 'Χ':'X', 'Ψ':'PS', 'Ω':'W',
|
bgneal@1
|
24 'Ά':'A', 'Έ':'E', 'Ί':'I', 'Ό':'O', 'Ύ':'Y', 'Ή':'H', 'Ώ':'W', 'Ϊ':'I',
|
bgneal@1
|
25 'Ϋ':'Y'
|
bgneal@1
|
26 }
|
bgneal@1
|
27 var TURKISH_MAP = {
|
bgneal@1
|
28 'ş':'s', 'Ş':'S', 'ı':'i', 'İ':'I', 'ç':'c', 'Ç':'C', 'ü':'u', 'Ü':'U',
|
bgneal@1
|
29 'ö':'o', 'Ö':'O', 'ğ':'g', 'Ğ':'G'
|
bgneal@1
|
30 }
|
bgneal@1
|
31 var RUSSIAN_MAP = {
|
bgneal@1
|
32 'а':'a', 'б':'b', 'в':'v', 'г':'g', 'д':'d', 'е':'e', 'ё':'yo', 'ж':'zh',
|
bgneal@1
|
33 'з':'z', 'и':'i', 'й':'j', 'к':'k', 'л':'l', 'м':'m', 'н':'n', 'о':'o',
|
bgneal@1
|
34 'п':'p', 'р':'r', 'с':'s', 'т':'t', 'у':'u', 'ф':'f', 'х':'h', 'ц':'c',
|
bgneal@1
|
35 'ч':'ch', 'ш':'sh', 'щ':'sh', 'ъ':'', 'ы':'y', 'ь':'', 'э':'e', 'ю':'yu',
|
bgneal@1
|
36 'я':'ya',
|
bgneal@1
|
37 'А':'A', 'Б':'B', 'В':'V', 'Г':'G', 'Д':'D', 'Е':'E', 'Ё':'Yo', 'Ж':'Zh',
|
bgneal@1
|
38 'З':'Z', 'И':'I', 'Й':'J', 'К':'K', 'Л':'L', 'М':'M', 'Н':'N', 'О':'O',
|
bgneal@1
|
39 'П':'P', 'Р':'R', 'С':'S', 'Т':'T', 'У':'U', 'Ф':'F', 'Х':'H', 'Ц':'C',
|
bgneal@1
|
40 'Ч':'Ch', 'Ш':'Sh', 'Щ':'Sh', 'Ъ':'', 'Ы':'Y', 'Ь':'', 'Э':'E', 'Ю':'Yu',
|
bgneal@1
|
41 'Я':'Ya'
|
bgneal@1
|
42 }
|
bgneal@1
|
43 var UKRAINIAN_MAP = {
|
bgneal@1
|
44 'Є':'Ye', 'І':'I', 'Ї':'Yi', 'Ґ':'G', 'є':'ye', 'і':'i', 'ї':'yi', 'ґ':'g'
|
bgneal@1
|
45 }
|
bgneal@1
|
46 var CZECH_MAP = {
|
bgneal@1
|
47 'č':'c', 'ď':'d', 'ě':'e', 'ň': 'n', 'ř':'r', 'š':'s', 'ť':'t', 'ů':'u',
|
bgneal@1
|
48 'ž':'z', 'Č':'C', 'Ď':'D', 'Ě':'E', 'Ň': 'N', 'Ř':'R', 'Š':'S', 'Ť':'T',
|
bgneal@1
|
49 'Ů':'U', 'Ž':'Z'
|
bgneal@1
|
50 }
|
bgneal@1
|
51
|
bgneal@1
|
52 var POLISH_MAP = {
|
bgneal@1
|
53 'ą':'a', 'ć':'c', 'ę':'e', 'ł':'l', 'ń':'n', 'ó':'o', 'ś':'s', 'ź':'z',
|
bgneal@1
|
54 'ż':'z', 'Ą':'A', 'Ć':'C', 'Ę':'e', 'Ł':'L', 'Ń':'N', 'Ó':'o', 'Ś':'S',
|
bgneal@1
|
55 'Ź':'Z', 'Ż':'Z'
|
bgneal@1
|
56 }
|
bgneal@1
|
57
|
bgneal@1
|
58 var LATVIAN_MAP = {
|
bgneal@1
|
59 'ā':'a', 'č':'c', 'ē':'e', 'ģ':'g', 'ī':'i', 'ķ':'k', 'ļ':'l', 'ņ':'n',
|
bgneal@1
|
60 'š':'s', 'ū':'u', 'ž':'z', 'Ā':'A', 'Č':'C', 'Ē':'E', 'Ģ':'G', 'Ī':'i',
|
bgneal@1
|
61 'Ķ':'k', 'Ļ':'L', 'Ņ':'N', 'Š':'S', 'Ū':'u', 'Ž':'Z'
|
bgneal@1
|
62 }
|
bgneal@1
|
63
|
bgneal@1
|
64 var ALL_DOWNCODE_MAPS=new Array()
|
bgneal@1
|
65 ALL_DOWNCODE_MAPS[0]=LATIN_MAP
|
bgneal@1
|
66 ALL_DOWNCODE_MAPS[1]=LATIN_SYMBOLS_MAP
|
bgneal@1
|
67 ALL_DOWNCODE_MAPS[2]=GREEK_MAP
|
bgneal@1
|
68 ALL_DOWNCODE_MAPS[3]=TURKISH_MAP
|
bgneal@1
|
69 ALL_DOWNCODE_MAPS[4]=RUSSIAN_MAP
|
bgneal@1
|
70 ALL_DOWNCODE_MAPS[5]=UKRAINIAN_MAP
|
bgneal@1
|
71 ALL_DOWNCODE_MAPS[6]=CZECH_MAP
|
bgneal@1
|
72 ALL_DOWNCODE_MAPS[7]=POLISH_MAP
|
bgneal@1
|
73 ALL_DOWNCODE_MAPS[8]=LATVIAN_MAP
|
bgneal@1
|
74
|
bgneal@1
|
75 var Downcoder = new Object();
|
bgneal@1
|
76 Downcoder.Initialize = function()
|
bgneal@1
|
77 {
|
bgneal@1
|
78 if (Downcoder.map) // already made
|
bgneal@1
|
79 return ;
|
bgneal@1
|
80 Downcoder.map ={}
|
bgneal@1
|
81 Downcoder.chars = '' ;
|
bgneal@1
|
82 for(var i in ALL_DOWNCODE_MAPS)
|
bgneal@1
|
83 {
|
bgneal@1
|
84 var lookup = ALL_DOWNCODE_MAPS[i]
|
bgneal@1
|
85 for (var c in lookup)
|
bgneal@1
|
86 {
|
bgneal@1
|
87 Downcoder.map[c] = lookup[c] ;
|
bgneal@1
|
88 Downcoder.chars += c ;
|
bgneal@1
|
89 }
|
bgneal@1
|
90 }
|
bgneal@1
|
91 Downcoder.regex = new RegExp('[' + Downcoder.chars + ']|[^' + Downcoder.chars + ']+','g') ;
|
bgneal@1
|
92 }
|
bgneal@1
|
93
|
bgneal@1
|
94 downcode= function( slug )
|
bgneal@1
|
95 {
|
bgneal@1
|
96 Downcoder.Initialize() ;
|
bgneal@1
|
97 var downcoded =""
|
bgneal@1
|
98 var pieces = slug.match(Downcoder.regex);
|
bgneal@1
|
99 if(pieces)
|
bgneal@1
|
100 {
|
bgneal@1
|
101 for (var i = 0 ; i < pieces.length ; i++)
|
bgneal@1
|
102 {
|
bgneal@1
|
103 if (pieces[i].length == 1)
|
bgneal@1
|
104 {
|
bgneal@1
|
105 var mapped = Downcoder.map[pieces[i]] ;
|
bgneal@1
|
106 if (mapped != null)
|
bgneal@1
|
107 {
|
bgneal@1
|
108 downcoded+=mapped;
|
bgneal@1
|
109 continue ;
|
bgneal@1
|
110 }
|
bgneal@1
|
111 }
|
bgneal@1
|
112 downcoded+=pieces[i];
|
bgneal@1
|
113 }
|
bgneal@1
|
114 }
|
bgneal@1
|
115 else
|
bgneal@1
|
116 {
|
bgneal@1
|
117 downcoded = slug;
|
bgneal@1
|
118 }
|
bgneal@1
|
119 return downcoded;
|
bgneal@1
|
120 }
|
bgneal@1
|
121
|
bgneal@1
|
122
|
bgneal@1
|
123 function URLify(s, num_chars) {
|
bgneal@1
|
124 // changes, e.g., "Petty theft" to "petty_theft"
|
bgneal@1
|
125 // remove all these words from the string before urlifying
|
bgneal@1
|
126 s = downcode(s);
|
bgneal@1
|
127 removelist = ["a", "an", "as", "at", "before", "but", "by", "for", "from",
|
bgneal@1
|
128 "is", "in", "into", "like", "of", "off", "on", "onto", "per",
|
bgneal@1
|
129 "since", "than", "the", "this", "that", "to", "up", "via",
|
bgneal@1
|
130 "with"];
|
bgneal@1
|
131 r = new RegExp('\\b(' + removelist.join('|') + ')\\b', 'gi');
|
bgneal@1
|
132 s = s.replace(r, '');
|
bgneal@1
|
133 // if downcode doesn't hit, the char will be stripped here
|
bgneal@1
|
134 s = s.replace(/[^-\w\s]/g, ''); // remove unneeded chars
|
bgneal@1
|
135 s = s.replace(/^\s+|\s+$/g, ''); // trim leading/trailing spaces
|
bgneal@1
|
136 s = s.replace(/[-\s]+/g, '-'); // convert spaces to hyphens
|
bgneal@1
|
137 s = s.toLowerCase(); // convert to lowercase
|
bgneal@1
|
138 return s.substring(0, num_chars);// trim to first num_chars chars
|
bgneal@1
|
139 }
|
bgneal@1
|
140
|