diff --git a/Ghidra/Features/Base/src/main/java/ghidra/app/util/bin/format/elf/ElfSymbolNameUtils.java b/Ghidra/Features/Base/src/main/java/ghidra/app/util/bin/format/elf/ElfSymbolNameUtils.java new file mode 100644 index 0000000000..64f138b068 --- /dev/null +++ b/Ghidra/Features/Base/src/main/java/ghidra/app/util/bin/format/elf/ElfSymbolNameUtils.java @@ -0,0 +1,60 @@ +/* ### + * IP: GHIDRA + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package ghidra.app.util.bin.format.elf; + +import ghidra.program.model.symbol.SymbolUtilities; + +public class ElfSymbolNameUtils { + + /** + * Converts a string with possible invalid characters into a valid symbol string. + *

+ * See {@link #getBadElfSymbolStringCodePointReplacement(int, int)} + * + * @param str symbol string to fix, null ok + * @return original str instance if already valid, otherwise fixed value + */ + public static String replaceInvalidChars(String str) { + return SymbolUtilities.replaceInvalidChars(str, + ElfSymbolNameUtils::getBadElfSymbolStringCodePointReplacement); + } + + /** + * Returns a replacement value for any bad code points found in an Elf symbol string. + * + * @param index index of the bad code point in the original string + * @param cp the bad code point + * @return replacement value to use instead of the bad code point + */ + public static String getBadElfSymbolStringCodePointReplacement(int index, int cp) { + if (cp < 0x20) { + // Format as ^Control character for consistency with readelf + // will range between ^@ .. ^_ (0..31) + return "^%c".formatted('@' + cp); + } + else if (cp == 0x7F) { + // Format as ^? character for consistency with readelf + return "^?"; + } + else if (cp == ' ') { + return "_"; + } + else { + return null; // omit the bad codepoint that caused this callback to be invoked + } + } + +} diff --git a/Ghidra/Features/Base/src/main/java/ghidra/app/util/bin/format/elf/relocation/AbstractElfRelocationHandler.java b/Ghidra/Features/Base/src/main/java/ghidra/app/util/bin/format/elf/relocation/AbstractElfRelocationHandler.java index 0d4b18a7d2..a0d06dd15a 100644 --- a/Ghidra/Features/Base/src/main/java/ghidra/app/util/bin/format/elf/relocation/AbstractElfRelocationHandler.java +++ b/Ghidra/Features/Base/src/main/java/ghidra/app/util/bin/format/elf/relocation/AbstractElfRelocationHandler.java @@ -4,9 +4,9 @@ * Licensed under the Apache License, Version 2.0 (the "License"); * you may not use this file except in compliance with the License. * You may obtain a copy of the License at - * + * * http://www.apache.org/licenses/LICENSE-2.0 - * + * * Unless required by applicable law or agreed to in writing, software * distributed under the License is distributed on an "AS IS" BASIS, * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. @@ -18,8 +18,7 @@ package ghidra.app.util.bin.format.elf.relocation; import java.util.HashMap; import java.util.Map; -import ghidra.app.util.bin.format.elf.ElfRelocation; -import ghidra.app.util.bin.format.elf.ElfSymbol; +import ghidra.app.util.bin.format.elf.*; import ghidra.app.util.importer.MessageLog; import ghidra.program.model.address.Address; import ghidra.program.model.listing.BookmarkType; @@ -103,6 +102,7 @@ abstract public class AbstractElfRelocationHandler { - if (cp < 0x20) { - // Format as ^Control character for consistency with readelf - cp += 0x40; // get ASCII control character, starts with ^@ - escapedBuf.append('^'); - escapedBuf.appendCodePoint(cp); - } - else if (cp == 0x7F) { - // Format as ^? character for consistency with readelf - escapedBuf.append("^?"); - } - else { - // Assume valid code point - escapedBuf.appendCodePoint(cp); - } - }); - return escapedBuf.toString(); - } @Override public void setElfSymbolAddress(ElfSymbol elfSymbol, Address address) { diff --git a/Ghidra/Features/Base/src/test/java/ghidra/program/model/symbol/SymbolUtilitiesNamingTest.java b/Ghidra/Features/Base/src/test/java/ghidra/program/model/symbol/SymbolUtilitiesNamingTest.java new file mode 100644 index 0000000000..483586eff8 --- /dev/null +++ b/Ghidra/Features/Base/src/test/java/ghidra/program/model/symbol/SymbolUtilitiesNamingTest.java @@ -0,0 +1,92 @@ +/* ### + * IP: GHIDRA + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ +package ghidra.program.model.symbol; + +import static ghidra.program.model.symbol.SymbolUtilities.*; +import static org.junit.Assert.*; + +import org.junit.Test; + +import ghidra.util.StringUtilities; + +public class SymbolUtilitiesNamingTest { + + @Test + public void testGoodStringObjectPassthru() { + String s = "testsym"; + assertSame(s, replaceInvalidChars(s, OMIT_BAD_CHARS)); + + s = "test sym"; + assertNotSame(s, replaceInvalidChars(s, OMIT_BAD_CHARS)); + } + + @Test + public void testNullString() { + assertNull(replaceInvalidChars(null, OMIT_BAD_CHARS)); + assertNull(SymbolUtilities.replaceInvalidChars(null, true)); + } + + @Test + public void testNullChar() { + assertEquals("testsym", replaceInvalidChars("test\0sym", OMIT_BAD_CHARS)); + assertTrue(SymbolUtilities.isInvalidCodePoint(0)); + } + + @Test + public void testBOMChar() { + assertEquals("testsym", + replaceInvalidChars( + "test" + Character.toString(StringUtilities.UNICODE_BE_BYTE_ORDER_MARK) + "sym", + OMIT_BAD_CHARS)); + } + + @Test + public void testRTLOChar() { + assertEquals("testsym", replaceInvalidChars("test\u202esym", OMIT_BAD_CHARS)); + } + + @Test + public void testBadCharRemoval() { + assertEquals("testsym", replaceInvalidChars("test sym", OMIT_BAD_CHARS)); + assertEquals("testsym", replaceInvalidChars("test\u007fsym", OMIT_BAD_CHARS)); + assertEquals("testsym", replaceInvalidChars("test\tsym", OMIT_BAD_CHARS)); + + assertEquals("test\uaabbsym", replaceInvalidChars("test\uaabbsym", OMIT_BAD_CHARS)); + } + + @Test + public void testBadCharReplaceWithUnderscores() { + assertEquals("test_sym", replaceInvalidChars("test sym", USE_UNDERSCORES)); + assertEquals("test_sym", replaceInvalidChars("test\u007fsym", USE_UNDERSCORES)); + assertEquals("test_sym", replaceInvalidChars("test\tsym", USE_UNDERSCORES)); + } + + @Test + public void testBadCharReplaceWithCustom() { + assertEquals("test_4_sym", + replaceInvalidChars("test sym", (index, cp) -> "_" + index + "_")); + } + + @Test + public void testAsciiRange() { + assertEquals( + "!\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~", + replaceInvalidChars( + "!\"#$%&'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~", + OMIT_BAD_CHARS)); + } + +} diff --git a/Ghidra/Features/GnuDemangler/ghidra_scripts/VxWorksSymTab_Finder.java b/Ghidra/Features/GnuDemangler/ghidra_scripts/VxWorksSymTab_Finder.java index 0ffcf0206e..49e8f910d5 100644 --- a/Ghidra/Features/GnuDemangler/ghidra_scripts/VxWorksSymTab_Finder.java +++ b/Ghidra/Features/GnuDemangler/ghidra_scripts/VxWorksSymTab_Finder.java @@ -41,8 +41,6 @@ // // @category VxWorks -import java.util.List; - import ghidra.app.cmd.data.CreateDataCmd; import ghidra.app.cmd.disassemble.DisassembleCommand; import ghidra.app.cmd.label.DemanglerCmd; @@ -57,7 +55,6 @@ import ghidra.program.model.data.*; import ghidra.program.model.listing.*; import ghidra.program.model.mem.MemoryBlock; import ghidra.program.model.symbol.*; -import ghidra.program.model.util.CodeUnitInsertionException; public class VxWorksSymTab_Finder extends GhidraScript { @@ -323,7 +320,7 @@ public class VxWorksSymTab_Finder extends GhidraScript { return false; } - while (!SymbolUtilities.isInvalidChar((char) _byte) && _byte != 0x00) { + while (_byte != 0x00 && !SymbolUtilities.isInvalidCodePoint(Byte.toUnsignedInt(_byte))) { if (monitor.isCancelled()) { return false; diff --git a/Ghidra/Framework/SoftwareModeling/src/main/java/ghidra/program/model/symbol/SymbolUtilities.java b/Ghidra/Framework/SoftwareModeling/src/main/java/ghidra/program/model/symbol/SymbolUtilities.java index 76975364c8..edc9d5c6fb 100644 --- a/Ghidra/Framework/SoftwareModeling/src/main/java/ghidra/program/model/symbol/SymbolUtilities.java +++ b/Ghidra/Framework/SoftwareModeling/src/main/java/ghidra/program/model/symbol/SymbolUtilities.java @@ -98,11 +98,6 @@ public class SymbolUtilities { */ public final static String ORDINAL_PREFIX = "Ordinal_"; - /** - * Invalid characters for a symbol name. - */ - public final static char[] INVALIDCHARS = { ' ' }; - private static final Comparator CASE_INSENSITIVE_SYMBOL_NAME_COMPARATOR = (s1, s2) -> { return s1.getName().compareToIgnoreCase(s2.getName()); }; @@ -135,24 +130,23 @@ public class SymbolUtilities { } /** - * Check for invalid characters - * (space or unprintable ascii below 0x20) - * in labels. + * Checks a string for invalid characters (control chars, space chars, whitespace chars). * * @param str the string to be checked for invalid characters. - * @return boolean true if no invalid chars + * @return boolean true if the string has invalid chars, false if string is valid */ public static boolean containsInvalidChars(String str) { - int len = str.length(); - for (int i = 0; i < len; i++) { - char c = str.charAt(i); - if (isInvalidChar(c)) { + for (int i = 0; i < str.length();) { + int codePoint = str.codePointAt(i); + if (isInvalidCodePoint(codePoint)) { return true; } + i += Character.charCount(codePoint); } return false; } + /** * Generates a default function name for a given address. * @param addr the entry point of the function. @@ -329,54 +323,135 @@ public class SymbolUtilities { } /** - * Returns true if the specified char - * is not valid for use in a symbol name + * Returns true if the specified char is not valid for use in a symbol name. + *

+ * See {@link #isInvalidCodePoint(int)} for better method that uses code points instead of + * chars. + * * @param c the character to be tested as a valid symbol character. - * @return return true if c is an invalid char within a symbol name, else false + * @return boolean true if c is an invalid char within a symbol name, else false */ public static boolean isInvalidChar(char c) { - if (c < ' ') { // non-printable ASCII - return true; - } + return isInvalidCodePoint(c); + } - for (char element : INVALIDCHARS) { - if (c == element) { + /** + * Returns true if the specified code point is not valid for use in a symbol name. + * + * @param cp the code point to be tested as a valid symbol character. + * @return boolean true if the code point is an invalid character within a symbol name, + * else false + */ + public static boolean isInvalidCodePoint(int cp) { + // Invisible / unprintable / whitespace unicode character categories. + // This bad list + the good list in the following comment are an exhaustive list of all + // unicode categories + switch (Character.getType(cp)) { + case Character.SPACE_SEPARATOR: + case Character.COMBINING_SPACING_MARK: + case Character.CONTROL: + case Character.ENCLOSING_MARK: + case Character.FORMAT: + case Character.LINE_SEPARATOR: + case Character.NON_SPACING_MARK: + case Character.PARAGRAPH_SEPARATOR: + case Character.PRIVATE_USE: + case Character.SURROGATE: + case Character.UNASSIGNED: return true; - } + /* + Unicode Character categories that are allowed: + Character.UPPERCASE_LETTER + Character.LOWERCASE_LETTER + Character.TITLECASE_LETTER + Character.MODIFIER_LETTER + Character.OTHER_LETTER + Character.DECIMAL_DIGIT_NUMBER + Character.LETTER_NUMBER + Character.OTHER_NUMBER + Character.DASH_PUNCTUATION + Character.START_PUNCTUATION + Character.END_PUNCTUATION + Character.CONNECTOR_PUNCTUATION + Character.OTHER_PUNCTUATION + Character.MATH_SYMBOL + Character.CURRENCY_SYMBOL + Character.MODIFIER_SYMBOL + Character.OTHER_SYMBOL + Character.INITIAL_QUOTE_PUNCTUATION + Character.FINAL_QUOTE_PUNCTUATION + */ } return false; } /** - * Removes from the given string any invalid characters or replaces - * them with underscores. - * - * For example: - * given "a:b*c", the return value would be "a_b_c" - * - * @param str the string to have invalid chars converted to underscores or removed. - * @param replaceWithUnderscore - true means replace the invalid - * chars with underscore. if false, then just drop the invalid chars - * @return modified string + * Callback functional interface, called by + * {@link SymbolUtilities#replaceInvalidChars(String, BadCharFixupFunc)} when it encounters a + * bad code point that needs addressing. (good characters in a string are NOT sent to + * this method) + */ + public interface BadCharFixupFunc { + String fixBadChar(int origIndex, int badCodePoint); + } + + /** + * BadCharFixupFunc that replaces bad characters with '_' underscores + */ + public static final BadCharFixupFunc USE_UNDERSCORES = (i, cp) -> "_"; + /** + * BadCharFixupFunc that removes bad characters from the string + */ + public static final BadCharFixupFunc OMIT_BAD_CHARS = (i, cp) -> null; + + /** + * Converts a string with possible invalid characters into a valid symbol string. + * + * @param str String to fix, {@code null} ok + * @param replaceWithUnderscore - true means replace the invalid chars with underscores, else + * if false, then just drop the invalid chars + * @return either the original String instance if already valid (or {@code null}), or a new + * string that contains the valid portions of the original with any bad chars removed or + * replaced with underscores. */ public static String replaceInvalidChars(String str, boolean replaceWithUnderscore) { + return replaceInvalidChars(str, replaceWithUnderscore ? USE_UNDERSCORES : OMIT_BAD_CHARS); + } + + /** + * Converts a string with possible invalid characters into a valid symbol string. + * + * @param str String to fix, {@code null} ok + * @param badCharFixup callback that controls how each bad char is fixed. It should return + * a string that should be used in place of the invalid character, or {@code null} if nothing + * should be used. + * @return either the original String instance if already valid (or {@code null}), or a new + * string that contains the valid portions of the original with any fixed-ups as returned by + * the badCharFixup callback. + */ + public static String replaceInvalidChars(String str, BadCharFixupFunc badCharFixup) { if (str == null) { return null; } - int len = str.length(); - StringBuilder buf = new StringBuilder(len); - for (int i = 0; i < len; ++i) { - char c = str.charAt(i); - if (isInvalidChar(c)) { - if (replaceWithUnderscore) { - buf.append(UNDERSCORE); + StringBuilder result = null; + for (int i = 0; i < str.length();) { + int codePoint = str.codePointAt(i); + if (isInvalidCodePoint(codePoint)) { + if (result == null) { + result = new StringBuilder(str.length()); + result.append(str.substring(0, i)); + } + String replacement = badCharFixup.fixBadChar(i, codePoint); + if (replacement != null) { + result.append(replacement); } } - else { - buf.append(c); + else if (result != null) { + result.appendCodePoint(codePoint); } + i += Character.charCount(codePoint); } - return buf.toString(); + return result != null ? result.toString() : str; } /**