Remove the regular space that you have first in the pattern:
str = str.replace(/[\u00A0\u1680โ\u180e\u2000-\u2009\u200aโ\u200bโ\u202f\u205fโ\u3000]/g,'');
Answer from Guffa on Stack OverflowRemove the regular space that you have first in the pattern:
str = str.replace(/[\u00A0\u1680โ\u180e\u2000-\u2009\u200aโ\u200bโ\u202f\u205fโ\u3000]/g,'');
try this:
var str = "Hello this is a test โof the site";
str= str.replace(/[\u00A0\u1680โ\u180e\u2000-\u2009\u200aโ\u200bโ\u202f\u205fโ\u3000]/g,'')
same as you did, but with out ' ' (regular space)
Efficient way of replacing special characters - JavaScript - SitePoint Forums | Web Development & Design Community
Replace unicode characters with characters (Javascript) - Stack Overflow
Javascript Replace Unicode Characters - JavaScript - SitePoint Forums | Web Development & Design Community
regex - Replace unicode matches in javascript - Stack Overflow
Hey, so I should start by saying, I donโt know if the title is actually what Iโm trying to do... hence my problem.
Iโm getting data from a wordpress api and the titles have, what I believe are Unicode or ASCII codes for symbols in them, โ&โ and โ-โ for example are a string of numbers, within a string that makes up the title.
Iโm not familiar enough with the wordpress api to know how to get it without that if thatโs even possible and Iโve tried many different ways to just change these codes within the string into their symbols... please... someone put me out my misery and tell me how to do this?
I donโt seem to be able to google the correct thing, trust me, Iโve tried! I think itโs because Iโm not phrasing the question correctly, but I donโt know how to phrase it :S
Cheers for any help ๐๐ผ
Edit - Iโm using React Native. Would decodeURIComponent(); be an option? Iโm not looking to work with a uri but it seems to have the effect Iโm after from looking at some google results :S
Adapted from Semplice, following link from here.
[^\x00-\x80] matches any character not in the ASCII range.
Note that some of the characters may not be encoded correctly from the copy and paste.
var latin_map = {"ร":"A","ฤ":"A","แบฎ":"A","แบถ":"A","แบฐ":"A","แบฒ":"A","แบด":"A","ว":"A","ร":"A","แบค":"A","แบฌ":"A","แบฆ":"A","แบจ":"A","แบช":"A","ร":"A","ว":"A","ศฆ":"A","ว ":"A","แบ ":"A","ศ":"A","ร":"A","แบข":"A","ศ":"A","ฤ":"A","ฤ":"A","ร
":"A","วบ":"A","แธ":"A","ศบ":"A","ร":"A","๊ฒ":"AA","ร":"AE","วผ":"AE","วข":"AE","๊ด":"AO","๊ถ":"AU","๊ธ":"AV","๊บ":"AV","๊ผ":"AY","แธ":"B","แธ":"B","ฦ":"B","แธ":"B","ษ":"B","ฦ":"B","ฤ":"C","ฤ":"C","ร":"C","แธ":"C","ฤ":"C","ฤ":"C","ฦ":"C","ศป":"C","ฤ":"D","แธ":"D","แธ":"D","แธ":"D","แธ":"D","ฦ":"D","แธ":"D","วฒ":"D","ว
":"D","ฤ":"D","ฦ":"D","วฑ":"DZ","ว":"DZ","ร":"E","ฤ":"E","ฤ":"E","ศจ":"E","แธ":"E","ร":"E","แบพ":"E","แป":"E","แป":"E","แป":"E","แป":"E","แธ":"E","ร":"E","ฤ":"E","แบธ":"E","ศ":"E","ร":"E","แบบ":"E","ศ":"E","ฤ":"E","แธ":"E","แธ":"E","ฤ":"E","ษ":"E","แบผ":"E","แธ":"E","๊ช":"ET","แธ":"F","ฦ":"F","วด":"G","ฤ":"G","วฆ":"G","ฤข":"G","ฤ":"G","ฤ ":"G","ฦ":"G","แธ ":"G","วค":"G","แธช":"H","ศ":"H","แธจ":"H","ฤค":"H","โฑง":"H","แธฆ":"H","แธข":"H","แธค":"H","ฤฆ":"H","ร":"I","ฤฌ":"I","ว":"I","ร":"I","ร":"I","แธฎ":"I","ฤฐ":"I","แป":"I","ศ":"I","ร":"I","แป":"I","ศ":"I","ฤช":"I","ฤฎ":"I","ฦ":"I","ฤจ":"I","แธฌ":"I","๊น":"D","๊ป":"F","๊ฝ":"G","๊":"R","๊":"S","๊":"T","๊ฌ":"IS","ฤด":"J","ษ":"J","แธฐ":"K","วจ":"K","ฤถ":"K","โฑฉ":"K","๊":"K","แธฒ":"K","ฦ":"K","แธด":"K","๊":"K","๊":"K","ฤน":"L","ศฝ":"L","ฤฝ":"L","ฤป":"L","แธผ":"L","แธถ":"L","แธธ":"L","โฑ ":"L","๊":"L","แธบ":"L","ฤฟ":"L","โฑข":"L","ว":"L","ล":"L","ว":"LJ","แธพ":"M","แน":"M","แน":"M","โฑฎ":"M","ล":"N","ล":"N","ล
":"N","แน":"N","แน":"N","แน":"N","วธ":"N","ฦ":"N","แน":"N","ศ ":"N","ว":"N","ร":"N","ว":"NJ","ร":"O","ล":"O","ว":"O","ร":"O","แป":"O","แป":"O","แป":"O","แป":"O","แป":"O","ร":"O","ศช":"O","ศฎ":"O","ศฐ":"O","แป":"O","ล":"O","ศ":"O","ร":"O","แป":"O","ฦ ":"O","แป":"O","แปข":"O","แป":"O","แป":"O","แป ":"O","ศ":"O","๊":"O","๊":"O","ล":"O","แน":"O","แน":"O","ฦ":"O","วช":"O","วฌ":"O","ร":"O","วพ":"O","ร":"O","แน":"O","แน":"O","ศฌ":"O","ฦข":"OI","๊":"OO","ฦ":"E","ฦ":"O","ศข":"OU","แน":"P","แน":"P","๊":"P","ฦค":"P","๊":"P","โฑฃ":"P","๊":"P","๊":"Q","๊":"Q","ล":"R","ล":"R","ล":"R","แน":"R","แน":"R","แน":"R","ศ":"R","ศ":"R","แน":"R","ษ":"R","โฑค":"R","๊พ":"C","ฦ":"E","ล":"S","แนค":"S","ล ":"S","แนฆ":"S","ล":"S","ล":"S","ศ":"S","แน ":"S","แนข":"S","แนจ":"S","ลค":"T","ลข":"T","แนฐ":"T","ศ":"T","ศพ":"T","แนช":"T","แนฌ":"T","ฦฌ":"T","แนฎ":"T","ฦฎ":"T","ลฆ":"T","โฑฏ":"A","๊":"L","ฦ":"M","ษ
":"V","๊จ":"TZ","ร":"U","ลฌ":"U","ว":"U","ร":"U","แนถ":"U","ร":"U","ว":"U","ว":"U","ว":"U","ว":"U","แนฒ":"U","แปค":"U","ลฐ":"U","ศ":"U","ร":"U","แปฆ":"U","ฦฏ":"U","แปจ":"U","แปฐ":"U","แปช":"U","แปฌ":"U","แปฎ":"U","ศ":"U","ลช":"U","แนบ":"U","ลฒ":"U","ลฎ":"U","ลจ":"U","แนธ":"U","แนด":"U","๊":"V","แนพ":"V","ฦฒ":"V","แนผ":"V","๊ ":"VY","แบ":"W","ลด":"W","แบ":"W","แบ":"W","แบ":"W","แบ":"W","โฑฒ":"W","แบ":"X","แบ":"X","ร":"Y","ลถ":"Y","ลธ":"Y","แบ":"Y","แปด":"Y","แปฒ":"Y","ฦณ":"Y","แปถ":"Y","แปพ":"Y","ศฒ":"Y","ษ":"Y","แปธ":"Y","ลน":"Z","ลฝ":"Z","แบ":"Z","โฑซ":"Z","ลป":"Z","แบ":"Z","ศค":"Z","แบ":"Z","ฦต":"Z","ฤฒ":"IJ","ล":"OE","แด":"A","แด":"AE","ส":"B","แด":"B","แด":"C","แด
":"D","แด":"E","๊ฐ":"F","ษข":"G","ส":"G","ส":"H","ษช":"I","ส":"R","แด":"J","แด":"K","ส":"L","แด":"L","แด":"M","ษด":"N","แด":"O","ษถ":"OE","แด":"O","แด":"OU","แด":"P","ส":"R","แด":"N","แด":"R","๊ฑ":"S","แด":"T","โฑป":"E","แด":"R","แด":"U","แด ":"V","แดก":"W","ส":"Y","แดข":"Z","รก":"a","ฤ":"a","แบฏ":"a","แบท":"a","แบฑ":"a","แบณ":"a","แบต":"a","ว":"a","รข":"a","แบฅ":"a","แบญ":"a","แบง":"a","แบฉ":"a","แบซ":"a","รค":"a","ว":"a","ศง":"a","วก":"a","แบก":"a","ศ":"a","ร ":"a","แบฃ":"a","ศ":"a","ฤ":"a","ฤ
":"a","แถ":"a","แบ":"a","รฅ":"a","วป":"a","แธ":"a","โฑฅ":"a","รฃ":"a","๊ณ":"aa","รฆ":"ae","วฝ":"ae","วฃ":"ae","๊ต":"ao","๊ท":"au","๊น":"av","๊ป":"av","๊ฝ":"ay","แธ":"b","แธ
":"b","ษ":"b","แธ":"b","แตฌ":"b","แถ":"b","ฦ":"b","ฦ":"b","ษต":"o","ฤ":"c","ฤ":"c","รง":"c","แธ":"c","ฤ":"c","ษ":"c","ฤ":"c","ฦ":"c","ศผ":"c","ฤ":"d","แธ":"d","แธ":"d","ศก":"d","แธ":"d","แธ":"d","ษ":"d","แถ":"d","แธ":"d","แตญ":"d","แถ":"d","ฤ":"d","ษ":"d","ฦ":"d","ฤฑ":"i","ศท":"j","ษ":"j","ส":"j","วณ":"dz","ว":"dz","รฉ":"e","ฤ":"e","ฤ":"e","ศฉ":"e","แธ":"e","รช":"e","แบฟ":"e","แป":"e","แป":"e","แป":"e","แป
":"e","แธ":"e","รซ":"e","ฤ":"e","แบน":"e","ศ
":"e","รจ":"e","แบป":"e","ศ":"e","ฤ":"e","แธ":"e","แธ":"e","โฑธ":"e","ฤ":"e","แถ":"e","ษ":"e","แบฝ":"e","แธ":"e","๊ซ":"et","แธ":"f","ฦ":"f","แตฎ":"f","แถ":"f","วต":"g","ฤ":"g","วง":"g","ฤฃ":"g","ฤ":"g","ฤก":"g","ษ ":"g","แธก":"g","แถ":"g","วฅ":"g","แธซ":"h","ศ":"h","แธฉ":"h","ฤฅ":"h","โฑจ":"h","แธง":"h","แธฃ":"h","แธฅ":"h","ษฆ":"h","แบ":"h","ฤง":"h","ฦ":"hv","รญ":"i","ฤญ":"i","ว":"i","รฎ":"i","รฏ":"i","แธฏ":"i","แป":"i","ศ":"i","รฌ":"i","แป":"i","ศ":"i","ฤซ":"i","ฤฏ":"i","แถ":"i","ษจ":"i","ฤฉ":"i","แธญ":"i","๊บ":"d","๊ผ":"f","แตน":"g","๊":"r","๊
":"s","๊":"t","๊ญ":"is","วฐ":"j","ฤต":"j","ส":"j","ษ":"j","แธฑ":"k","วฉ":"k","ฤท":"k","โฑช":"k","๊":"k","แธณ":"k","ฦ":"k","แธต":"k","แถ":"k","๊":"k","๊
":"k","ฤบ":"l","ฦ":"l","ษฌ":"l","ฤพ":"l","ฤผ":"l","แธฝ":"l","ศด":"l","แธท":"l","แธน":"l","โฑก":"l","๊":"l","แธป":"l","ล":"l","ษซ":"l","แถ
":"l","ษญ":"l","ล":"l","ว":"lj","ลฟ":"s","แบ":"s","แบ":"s","แบ":"s","แธฟ":"m","แน":"m","แน":"m","ษฑ":"m","แตฏ":"m","แถ":"m","ล":"n","ล":"n","ล":"n","แน":"n","ศต":"n","แน
":"n","แน":"n","วน":"n","ษฒ":"n","แน":"n","ฦ":"n","แตฐ":"n","แถ":"n","ษณ":"n","รฑ":"n","ว":"nj","รณ":"o","ล":"o","ว":"o","รด":"o","แป":"o","แป":"o","แป":"o","แป":"o","แป":"o","รถ":"o","ศซ":"o","ศฏ":"o","ศฑ":"o","แป":"o","ล":"o","ศ":"o","รฒ":"o","แป":"o","ฦก":"o","แป":"o","แปฃ":"o","แป":"o","แป":"o","แปก":"o","ศ":"o","๊":"o","๊":"o","โฑบ":"o","ล":"o","แน":"o","แน":"o","วซ":"o","วญ":"o","รธ":"o","วฟ":"o","รต":"o","แน":"o","แน":"o","ศญ":"o","ฦฃ":"oi","๊":"oo","ษ":"e","แถ":"e","ษ":"o","แถ":"o","ศฃ":"ou","แน":"p","แน":"p","๊":"p","ฦฅ":"p","แตฑ":"p","แถ":"p","๊":"p","แตฝ":"p","๊":"p","๊":"q","ส ":"q","ษ":"q","๊":"q","ล":"r","ล":"r","ล":"r","แน":"r","แน":"r","แน":"r","ศ":"r","ษพ":"r","แตณ":"r","ศ":"r","แน":"r","ษผ":"r","แตฒ":"r","แถ":"r","ษ":"r","ษฝ":"r","โ":"c","๊ฟ":"c","ษ":"e","ษฟ":"r","ล":"s","แนฅ":"s","ลก":"s","แนง":"s","ล":"s","ล":"s","ศ":"s","แนก":"s","แนฃ":"s","แนฉ":"s","ส":"s","แตด":"s","แถ":"s","ศฟ":"s","ษก":"g","แด":"o","แด":"o","แด":"u","ลฅ":"t","ลฃ":"t","แนฑ":"t","ศ":"t","ศถ":"t","แบ":"t","โฑฆ":"t","แนซ":"t","แนญ":"t","ฦญ":"t","แนฏ":"t","แตต":"t","ฦซ":"t","ส":"t","ลง":"t","แตบ":"th","ษ":"a","แด":"ae","ว":"e","แตท":"g","ษฅ":"h","สฎ":"h","สฏ":"h","แด":"i","ส":"k","๊":"l","ษฏ":"m","ษฐ":"m","แด":"oe","ษน":"r","ษป":"r","ษบ":"r","โฑน":"r","ส":"t","ส":"v","ส":"w","ส":"y","๊ฉ":"tz","รบ":"u","ลญ":"u","ว":"u","รป":"u","แนท":"u","รผ":"u","ว":"u","ว":"u","ว":"u","ว":"u","แนณ":"u","แปฅ":"u","ลฑ":"u","ศ":"u","รน":"u","แปง":"u","ฦฐ":"u","แปฉ":"u","แปฑ":"u","แปซ":"u","แปญ":"u","แปฏ":"u","ศ":"u","ลซ":"u","แนป":"u","ลณ":"u","แถ":"u","ลฏ":"u","ลฉ":"u","แนน":"u","แนต":"u","แตซ":"ue","๊ธ":"um","โฑด":"v","๊":"v","แนฟ":"v","ส":"v","แถ":"v","โฑฑ":"v","แนฝ":"v","๊ก":"vy","แบ":"w","ลต":"w","แบ
":"w","แบ":"w","แบ":"w","แบ":"w","โฑณ":"w","แบ":"w","แบ":"x","แบ":"x","แถ":"x","รฝ":"y","ลท":"y","รฟ":"y","แบ":"y","แปต":"y","แปณ":"y","ฦด":"y","แปท":"y","แปฟ":"y","ศณ":"y","แบ":"y","ษ":"y","แปน":"y","ลบ":"z","ลพ":"z","แบ":"z","ส":"z","โฑฌ":"z","ลผ":"z","แบ":"z","ศฅ":"z","แบ":"z","แตถ":"z","แถ":"z","ส":"z","ฦถ":"z","ษ":"z","๏ฌ":"ff","๏ฌ":"ffi","๏ฌ":"ffl","๏ฌ":"fi","๏ฌ":"fl","ฤณ":"ij","ล":"oe","๏ฌ":"st","โ":"a","โ":"e","แตข":"i","โฑผ":"j","โ":"o","แตฃ":"r","แตค":"u","แตฅ":"v","โ":"x"};
function embolden( str, chr ){
return str.replace( /[^\x00-\x80]/g,
function (a) {
return chr == latin_map[a] ? '<b>' + a + '</b>' : a;
}
);
}
embolden( 'รกdรกm', 'a' ); // "<b>รก</b>d<b>รก</b>m"
I've tried this code, see if it's what you're looking for:
'รกdรกm'.replace(/./g,function(char){
switch(char.toLowerCase()){
case 'รก':
case 'ร ':
case 'รข':
case 'รฃ':
return '*';
break;
}
return char;
});
EDIT:
To replace all chars that don't belong to the ASCII table, just check if the char has a char code up to 127, since the ASCII table char codes are defined between 0 and 127 (notice that รก doesn't belong to the Unicode table, but to the Extended ASCII table, that comes from 0 up to 255):
'รกdรกm'.replace(/./g,function(char){
return char.charCodeAt(0)<=127 ? char : '<b>' + char + '</b>';
});
By using Regx. in string replace you can achieve this. See the below code
var unicode_dictionary = {
"\\00E9": "รฉ",
"\\00E0": "ร "
}
var old_str = "rapport couvrant une p\00E9riode de 6 mois (f\00E9vrier \00E0 juillet)"
function convert(){
for(var key in unicode_dictionary){
var regx=new RegExp(key,'g')
old_str=old_str.replace(regx,unicode_dictionary[key]);
}
alert(old_str);
}
<script src="https://ajax.googleapis.com/ajax/libs/jquery/1.9.1/jquery.min.js"></script>
<button onclick='convert()'>Convert</button>
Run code snippetEdit code snippet Hide Results Copy to answer Expand
You can use a Regex in combination with its function capability: search for the pattern \\[hex digits] and replace it with the actual Unicode character, in one pass, for any code. As long as these codes represent valid Unicode characters, the following works:
var old_str = "rapport couvrant une p\\00E9riode de 6 mois (f\\00E9vrier \\00E0 juillet)";
var new_str = old_str.replace(/\\([\da-f]{4})/gi, function (a,b)
{
return String.fromCharCode(parseInt(b, 16));
});
Note that I doubled the backslashes in the source string of the snippet because that is per Javascript rules. The single backslashes in your source text do not need this.
This parses exactly 4 hexadecimal characters. If there may be less but no more than 4, you can use the regex \\([\da-f]{1,4}). It needs a maximum limit because there is no end marker in the source sequence. That means that without the maximum of 4, a string such as
the number \\00224\\0022
-- intended the number "4" -- will be translated as
the number ศค"
because the Unicode codepoint U+0224 represents a capital Z with hook.
Sure is!
Running this in the Firebug console
"ยฎรผ".replace(/[ยฎรผ]/g,"replaced")
returned
"replacedreplaced"
You can also do
"ยฎรผ".replace(/[\xAE\xFC]/g,"Wohoo! ");
which returns
"Wohoo! Wohoo! "
A good hex symbol lookup page can be found at http://www.ascii.cl/htmlcodes.htm
Example
running this jQuery on this page
$(".post-text").text().replace(/ยฎ/g," ******** ")
returns
" is it possible to translate special characters like ******** , รผ etc with javascript
String replace function? Use this syntax... string.replace(/\xCC/g/,''); Where 'CC' is
the hex character code for the char you are wanting to replace. In this example I am
replacing with empty string ''. yes, and is as simple as can be: ' ******** '.replace('
******** ','anything'); Sure is! Running this in the Firebug console " ******** รผ".
replace(/[ ******** รผ]/g,"replaced") returned replacedreplaced "
yes, and is as simple as can be:
'ยฎ'.replace('ยฎ','anything');
RegEx Improvements
The regex can be shortened by using case-insensitive match with i flag. We can remove the characters which are added as both lowercase and uppercase in the regex.
After removing lowercase characters regex will be as below
\?รร_|ล |ลฝ|ร|ร|ร|ร|ร|ร
|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|ร|รฐ|รฟ|_ลโ|__|_
Here's live demo of regex
The regex can be further improved by using character class which will make the matches faster than OR conditions
\?รร_|_ลโ|[ล ลฝรรรรรร
รรรรรรรรรรรรรรรรรรรรรรรรรฐรฟ_]+
Adding + quantifier also has positive effect on the number of steps taken to match characters when the characters in the character class are consecutive/adjacent to each other.
Here's the demo on RegEx101, without + quantifierScreenshot and with + quantifierscreenshot applied on the same data. Note that in these demos, PHP is selected as the steps taken to match is not shown for JavaScript. Also, the regex is different, it also contains lowercase counterparts of those special characters as i flag is not working with PHP and don't want to apply u(Unicode) flag as it is not supported in JavaScript.
These demos are created only to show difference when + is applied on character class. The effect should be similar in JavaScript.
Note that the __(two underscores) are redundant as _ is already added in character class and with g flag it'll remove all occurrences.
Method Chaining
As replace returns a string, any other string method can be called on it. Multiple calls to replace can be chained.
str.replace(someRegexOrString, someString)
.replace(someOtherRegexOrString, someOtherString);
This is equivalent to
var temp = str.replace(someRegexOrString, someString);
var result = temp.replace(someOtherRegexOrString, someOtherString);
Replacing HTML
jQuery html() accepts a function which will receive the current innerHTML of the element on which the method is called as parameter and replaces the returned content to the element.
The code can be written as
$('.rte').html(function(index, currentHTML) {
return doSomeOperationOn(currentHTML);
});
Complete Code
With above changes, the code will be
$(document).ready(function() {
var regex = /\?รร_|_ลโ|[ล ลฝรรรรรร
รรรรรรรรรรรรรรรรรรรรรรรรรฐรฟ_]+/gi;
$('.rte').html(function(i, oldHTML) {
return oldHTML.replace(regex, ' ')
.replace(/[^\x00-\x7F]|\?/g, '');
});
});
$(document).ready(function() { is more readable than $(function() {. So, you may also consider using more expressive form.
The code may be correct in itself, but it does the wrong thing.
If by strange you mean unknown to someone who only knows English, that's no excuse for removing any letters you don't know. Would you really want to look at street signs for Cafs (which were legitimate Cafรฉs before)?
If you get strange character sequences like รยถ, that's an encoding problem and you need to fix it properly instead of hiding it.
If you really have to keep your code, at least be honest and replace each unknown character with a question mark or the Unicode replacement character so that it is clearly visible that something unexpected happened here.
HTML entities get converted to actual unicode characters. Example:
document.body.innerHTML = "⇈";
console.log(document.body.innerHTML === "\u21c8"); // true
// Instead of a unicode esape sequence, you can write the
// actual unicode character. This is safe as long as you
// specify the correct encoding for your JavaScript files:
console.log(document.body.innerHTML === "โ"); // true
Run code snippetEdit code snippet Hide Results Copy to answer Expand
So when you read the innerHTML via $(this).html() or the textContent via $(this).text(), you need to look for the actual unicode character given by its unicode escape sequence "\u21c8" or directly "โ" and not its entity "⇈".
Try using String.fromCharCode()
The static String.fromCharCode() method returns a string created by using the specified sequence of Unicode values.
Syntax
String.fromCharCode(num1[, ...[, numN]]);
Examples
myString.replace(String.fromCharCode(8648),''); for โ
myString.replace(String.fromCharCode(8650),''); for โ
myString.indexOf(String.fromCharCode(84,69,83,84);
ยป npm install replace-special-characters