1 //===-- StringPrinter.cpp ----------------------------------------*- C++ -*-===//
3 // The LLVM Compiler Infrastructure
5 // This file is distributed under the University of Illinois Open Source
6 // License. See LICENSE.TXT for details.
8 //===----------------------------------------------------------------------===//
10 #include "lldb/DataFormatters/StringPrinter.h"
12 #include "lldb/Core/Debugger.h"
13 #include "lldb/Core/Error.h"
14 #include "lldb/Core/ValueObject.h"
15 #include "lldb/Target/Language.h"
16 #include "lldb/Target/Process.h"
17 #include "lldb/Target/Target.h"
19 #include "llvm/Support/ConvertUTF.h"
25 using namespace lldb_private;
26 using namespace lldb_private::formatters;
28 // we define this for all values of type but only implement it for those we care about
29 // that's good because we get linker errors for any unsupported type
30 template <lldb_private::formatters::StringPrinter::StringElementType type>
31 static StringPrinter::StringPrinterBufferPointer<>
32 GetPrintableImpl(uint8_t* buffer, uint8_t* buffer_end, uint8_t*& next);
34 // mimic isprint() for Unicode codepoints
36 isprint(char32_t codepoint)
38 if (codepoint <= 0x1F || codepoint == 0x7F) // C0
42 if (codepoint >= 0x80 && codepoint <= 0x9F) // C1
46 if (codepoint == 0x2028 || codepoint == 0x2029) // line/paragraph separators
50 if (codepoint == 0x200E || codepoint == 0x200F || (codepoint >= 0x202A && codepoint <= 0x202E)) // bidirectional text control
54 if (codepoint >= 0xFFF9 && codepoint <= 0xFFFF) // interlinears and generally specials
62 StringPrinter::StringPrinterBufferPointer<>
63 GetPrintableImpl<StringPrinter::StringElementType::ASCII> (uint8_t* buffer, uint8_t* buffer_end, uint8_t*& next)
65 StringPrinter::StringPrinterBufferPointer<> retval = {nullptr};
100 if (isprint(*buffer))
104 uint8_t* data = new uint8_t[5];
105 sprintf((char*)data,"\\x%02x",*buffer);
106 retval = {data, 4, [] (const uint8_t* c) {delete[] c;} };
116 ConvertUTF8ToCodePoint (unsigned char c0, unsigned char c1)
118 return (c0-192)*64+(c1-128);
121 ConvertUTF8ToCodePoint (unsigned char c0, unsigned char c1, unsigned char c2)
123 return (c0-224)*4096+(c1-128)*64+(c2-128);
126 ConvertUTF8ToCodePoint (unsigned char c0, unsigned char c1, unsigned char c2, unsigned char c3)
128 return (c0-240)*262144+(c2-128)*4096+(c2-128)*64+(c3-128);
132 StringPrinter::StringPrinterBufferPointer<>
133 GetPrintableImpl<StringPrinter::StringElementType::UTF8> (uint8_t* buffer, uint8_t* buffer_end, uint8_t*& next)
135 StringPrinter::StringPrinterBufferPointer<> retval {nullptr};
137 unsigned utf8_encoded_len = getNumBytesForUTF8(*buffer);
139 if (1+buffer_end-buffer < utf8_encoded_len)
141 // I don't have enough bytes - print whatever I have left
142 retval = {buffer,static_cast<size_t>(1+buffer_end-buffer)};
147 char32_t codepoint = 0;
148 switch (utf8_encoded_len)
151 // this is just an ASCII byte - ask ASCII
152 return GetPrintableImpl<StringPrinter::StringElementType::ASCII>(buffer, buffer_end, next);
154 codepoint = ConvertUTF8ToCodePoint((unsigned char)*buffer, (unsigned char)*(buffer+1));
157 codepoint = ConvertUTF8ToCodePoint((unsigned char)*buffer, (unsigned char)*(buffer+1), (unsigned char)*(buffer+2));
160 codepoint = ConvertUTF8ToCodePoint((unsigned char)*buffer, (unsigned char)*(buffer+1), (unsigned char)*(buffer+2), (unsigned char)*(buffer+3));
163 // this is probably some bogus non-character thing
164 // just print it as-is and hope to sync up again soon
205 if (isprint(codepoint))
206 retval = {buffer,utf8_encoded_len};
209 uint8_t* data = new uint8_t[11];
210 sprintf((char *)data, "\\U%08x", (unsigned)codepoint);
211 retval = { data,10,[] (const uint8_t* c) {delete[] c;} };
216 next = buffer + utf8_encoded_len;
220 // this should not happen - but just in case.. try to resync at some point
226 // Given a sequence of bytes, this function returns:
227 // a sequence of bytes to actually print out + a length
228 // the following unscanned position of the buffer is in next
229 static StringPrinter::StringPrinterBufferPointer<>
230 GetPrintable(StringPrinter::StringElementType type, uint8_t* buffer, uint8_t* buffer_end, uint8_t*& next)
237 case StringPrinter::StringElementType::ASCII:
238 return GetPrintableImpl<StringPrinter::StringElementType::ASCII>(buffer, buffer_end, next);
239 case StringPrinter::StringElementType::UTF8:
240 return GetPrintableImpl<StringPrinter::StringElementType::UTF8>(buffer, buffer_end, next);
246 StringPrinter::EscapingHelper
247 StringPrinter::GetDefaultEscapingHelper (GetPrintableElementType elem_type)
251 case GetPrintableElementType::UTF8:
252 return [] (uint8_t* buffer, uint8_t* buffer_end, uint8_t*& next) -> StringPrinter::StringPrinterBufferPointer<> {
253 return GetPrintable(StringPrinter::StringElementType::UTF8, buffer, buffer_end, next);
255 case GetPrintableElementType::ASCII:
256 return [] (uint8_t* buffer, uint8_t* buffer_end, uint8_t*& next) -> StringPrinter::StringPrinterBufferPointer<> {
257 return GetPrintable(StringPrinter::StringElementType::ASCII, buffer, buffer_end, next);
260 llvm_unreachable("bad element type");
263 // use this call if you already have an LLDB-side buffer for the data
264 template<typename SourceDataType>
266 DumpUTFBufferToStream (ConversionResult (*ConvertFunction) (const SourceDataType**,
267 const SourceDataType*,
271 const StringPrinter::ReadBufferAndDumpToStreamOptions& dump_options)
273 Stream &stream(*dump_options.GetStream());
274 if (dump_options.GetPrefixToken() != 0)
275 stream.Printf("%s",dump_options.GetPrefixToken());
276 if (dump_options.GetQuote() != 0)
277 stream.Printf("%c",dump_options.GetQuote());
278 auto data(dump_options.GetData());
279 auto source_size(dump_options.GetSourceSize());
280 if (data.GetByteSize() && data.GetDataStart() && data.GetDataEnd())
282 const int bufferSPSize = data.GetByteSize();
283 if (dump_options.GetSourceSize() == 0)
285 const int origin_encoding = 8*sizeof(SourceDataType);
286 source_size = bufferSPSize/(origin_encoding / 4);
289 const SourceDataType *data_ptr = (const SourceDataType*)data.GetDataStart();
290 const SourceDataType *data_end_ptr = data_ptr + source_size;
292 const bool zero_is_terminator = dump_options.GetBinaryZeroIsTerminator();
294 if (zero_is_terminator)
296 while (data_ptr < data_end_ptr)
300 data_end_ptr = data_ptr;
306 data_ptr = (const SourceDataType*)data.GetDataStart();
309 lldb::DataBufferSP utf8_data_buffer_sp;
310 UTF8* utf8_data_ptr = nullptr;
311 UTF8* utf8_data_end_ptr = nullptr;
315 utf8_data_buffer_sp.reset(new DataBufferHeap(4*bufferSPSize,0));
316 utf8_data_ptr = (UTF8*)utf8_data_buffer_sp->GetBytes();
317 utf8_data_end_ptr = utf8_data_ptr + utf8_data_buffer_sp->GetByteSize();
318 ConvertFunction ( &data_ptr, data_end_ptr, &utf8_data_ptr, utf8_data_end_ptr, lenientConversion );
319 if (false == zero_is_terminator)
320 utf8_data_end_ptr = utf8_data_ptr;
321 utf8_data_ptr = (UTF8*)utf8_data_buffer_sp->GetBytes(); // needed because the ConvertFunction will change the value of the data_ptr
325 // just copy the pointers - the cast is necessary to make the compiler happy
326 // but this should only happen if we are reading UTF8 data
327 utf8_data_ptr = const_cast<UTF8 *>(reinterpret_cast<const UTF8*>(data_ptr));
328 utf8_data_end_ptr = const_cast<UTF8 *>(reinterpret_cast<const UTF8*>(data_end_ptr));
331 const bool escape_non_printables = dump_options.GetEscapeNonPrintables();
332 lldb_private::formatters::StringPrinter::EscapingHelper escaping_callback;
333 if (escape_non_printables)
335 if (Language *language = Language::FindPlugin(dump_options.GetLanguage()))
336 escaping_callback = language->GetStringPrinterEscapingHelper(lldb_private::formatters::StringPrinter::GetPrintableElementType::UTF8);
338 escaping_callback = lldb_private::formatters::StringPrinter::GetDefaultEscapingHelper(lldb_private::formatters::StringPrinter::GetPrintableElementType::UTF8);
341 // since we tend to accept partial data (and even partially malformed data)
342 // we might end up with no NULL terminator before the end_ptr
343 // hence we need to take a slower route and ensure we stay within boundaries
344 for (;utf8_data_ptr < utf8_data_end_ptr;)
346 if (zero_is_terminator && !*utf8_data_ptr)
349 if (escape_non_printables)
351 uint8_t* next_data = nullptr;
352 auto printable = escaping_callback(utf8_data_ptr, utf8_data_end_ptr, next_data);
353 auto printable_bytes = printable.GetBytes();
354 auto printable_size = printable.GetSize();
355 if (!printable_bytes || !next_data)
357 // GetPrintable() failed on us - print one byte in a desperate resync attempt
358 printable_bytes = utf8_data_ptr;
360 next_data = utf8_data_ptr+1;
362 for (unsigned c = 0; c < printable_size; c++)
363 stream.Printf("%c", *(printable_bytes+c));
364 utf8_data_ptr = (uint8_t*)next_data;
368 stream.Printf("%c",*utf8_data_ptr);
373 if (dump_options.GetQuote() != 0)
374 stream.Printf("%c",dump_options.GetQuote());
375 if (dump_options.GetSuffixToken() != 0)
376 stream.Printf("%s",dump_options.GetSuffixToken());
377 if (dump_options.GetIsTruncated())
378 stream.Printf("...");
382 lldb_private::formatters::StringPrinter::ReadStringAndDumpToStreamOptions::ReadStringAndDumpToStreamOptions (ValueObject& valobj) :
383 ReadStringAndDumpToStreamOptions()
385 SetEscapeNonPrintables(valobj.GetTargetSP()->GetDebugger().GetEscapeNonPrintables());
388 lldb_private::formatters::StringPrinter::ReadBufferAndDumpToStreamOptions::ReadBufferAndDumpToStreamOptions (ValueObject& valobj) :
389 ReadBufferAndDumpToStreamOptions()
391 SetEscapeNonPrintables(valobj.GetTargetSP()->GetDebugger().GetEscapeNonPrintables());
394 lldb_private::formatters::StringPrinter::ReadBufferAndDumpToStreamOptions::ReadBufferAndDumpToStreamOptions (const ReadStringAndDumpToStreamOptions& options) :
395 ReadBufferAndDumpToStreamOptions()
397 SetStream(options.GetStream());
398 SetPrefixToken(options.GetPrefixToken());
399 SetSuffixToken(options.GetSuffixToken());
400 SetQuote(options.GetQuote());
401 SetEscapeNonPrintables(options.GetEscapeNonPrintables());
402 SetBinaryZeroIsTerminator(options.GetBinaryZeroIsTerminator());
403 SetLanguage(options.GetLanguage());
407 namespace lldb_private
415 StringPrinter::ReadStringAndDumpToStream<StringPrinter::StringElementType::ASCII> (const ReadStringAndDumpToStreamOptions& options)
417 assert(options.GetStream() && "need a Stream to print the string to");
420 ProcessSP process_sp(options.GetProcessSP());
422 if (process_sp.get() == nullptr || options.GetLocation() == 0)
426 const auto max_size = process_sp->GetTarget().GetMaximumSizeOfStringSummary();
427 bool is_truncated = false;
429 if (options.GetSourceSize() == 0)
431 else if (!options.GetIgnoreMaxLength())
433 size = options.GetSourceSize();
441 size = options.GetSourceSize();
443 lldb::DataBufferSP buffer_sp(new DataBufferHeap(size,0));
445 process_sp->ReadCStringFromMemory(options.GetLocation(), (char*)buffer_sp->GetBytes(), size, my_error);
450 const char* prefix_token = options.GetPrefixToken();
451 char quote = options.GetQuote();
453 if (prefix_token != 0)
454 options.GetStream()->Printf("%s%c",prefix_token,quote);
456 options.GetStream()->Printf("%c",quote);
458 uint8_t* data_end = buffer_sp->GetBytes()+buffer_sp->GetByteSize();
460 const bool escape_non_printables = options.GetEscapeNonPrintables();
461 lldb_private::formatters::StringPrinter::EscapingHelper escaping_callback;
462 if (escape_non_printables)
464 if (Language *language = Language::FindPlugin(options.GetLanguage()))
465 escaping_callback = language->GetStringPrinterEscapingHelper(lldb_private::formatters::StringPrinter::GetPrintableElementType::ASCII);
467 escaping_callback = lldb_private::formatters::StringPrinter::GetDefaultEscapingHelper(lldb_private::formatters::StringPrinter::GetPrintableElementType::ASCII);
470 // since we tend to accept partial data (and even partially malformed data)
471 // we might end up with no NULL terminator before the end_ptr
472 // hence we need to take a slower route and ensure we stay within boundaries
473 for (uint8_t* data = buffer_sp->GetBytes(); *data && (data < data_end);)
475 if (escape_non_printables)
477 uint8_t* next_data = nullptr;
478 auto printable = escaping_callback(data, data_end, next_data);
479 auto printable_bytes = printable.GetBytes();
480 auto printable_size = printable.GetSize();
481 if (!printable_bytes || !next_data)
483 // GetPrintable() failed on us - print one byte in a desperate resync attempt
484 printable_bytes = data;
488 for (unsigned c = 0; c < printable_size; c++)
489 options.GetStream()->Printf("%c", *(printable_bytes+c));
490 data = (uint8_t*)next_data;
494 options.GetStream()->Printf("%c",*data);
499 const char* suffix_token = options.GetSuffixToken();
501 if (suffix_token != 0)
502 options.GetStream()->Printf("%c%s",quote, suffix_token);
504 options.GetStream()->Printf("%c",quote);
507 options.GetStream()->Printf("...");
512 template<typename SourceDataType>
514 ReadUTFBufferAndDumpToStream (const StringPrinter::ReadStringAndDumpToStreamOptions& options,
515 ConversionResult (*ConvertFunction) (const SourceDataType**,
516 const SourceDataType*,
521 assert(options.GetStream() && "need a Stream to print the string to");
523 if (options.GetLocation() == 0 || options.GetLocation() == LLDB_INVALID_ADDRESS)
526 lldb::ProcessSP process_sp(options.GetProcessSP());
531 const int type_width = sizeof(SourceDataType);
532 const int origin_encoding = 8 * type_width ;
533 if (origin_encoding != 8 && origin_encoding != 16 && origin_encoding != 32)
535 // if not UTF8, I need a conversion function to return proper UTF8
536 if (origin_encoding != 8 && !ConvertFunction)
539 if (!options.GetStream())
542 uint32_t sourceSize = options.GetSourceSize();
543 bool needs_zero_terminator = options.GetNeedsZeroTermination();
545 bool is_truncated = false;
546 const auto max_size = process_sp->GetTarget().GetMaximumSizeOfStringSummary();
550 sourceSize = max_size;
551 needs_zero_terminator = true;
553 else if (!options.GetIgnoreMaxLength())
555 if (sourceSize > max_size)
557 sourceSize = max_size;
562 const int bufferSPSize = sourceSize * type_width;
564 lldb::DataBufferSP buffer_sp(new DataBufferHeap(bufferSPSize,0));
566 if (!buffer_sp->GetBytes())
570 char *buffer = reinterpret_cast<char *>(buffer_sp->GetBytes());
572 if (needs_zero_terminator)
573 process_sp->ReadStringFromMemory(options.GetLocation(), buffer, bufferSPSize, error, type_width);
575 process_sp->ReadMemoryFromInferior(options.GetLocation(), (char*)buffer_sp->GetBytes(), bufferSPSize, error);
579 options.GetStream()->Printf("unable to read data");
583 DataExtractor data(buffer_sp, process_sp->GetByteOrder(), process_sp->GetAddressByteSize());
585 StringPrinter::ReadBufferAndDumpToStreamOptions dump_options(options);
586 dump_options.SetData(data);
587 dump_options.SetSourceSize(sourceSize);
588 dump_options.SetIsTruncated(is_truncated);
590 return DumpUTFBufferToStream(ConvertFunction, dump_options);
595 StringPrinter::ReadStringAndDumpToStream<StringPrinter::StringElementType::UTF8> (const ReadStringAndDumpToStreamOptions& options)
597 return ReadUTFBufferAndDumpToStream<UTF8>(options,
603 StringPrinter::ReadStringAndDumpToStream<StringPrinter::StringElementType::UTF16> (const ReadStringAndDumpToStreamOptions& options)
605 return ReadUTFBufferAndDumpToStream<UTF16>(options,
611 StringPrinter::ReadStringAndDumpToStream<StringPrinter::StringElementType::UTF32> (const ReadStringAndDumpToStreamOptions& options)
613 return ReadUTFBufferAndDumpToStream<UTF32>(options,
619 StringPrinter::ReadBufferAndDumpToStream<StringPrinter::StringElementType::UTF8> (const ReadBufferAndDumpToStreamOptions& options)
621 assert(options.GetStream() && "need a Stream to print the string to");
623 return DumpUTFBufferToStream<UTF8>(nullptr, options);
628 StringPrinter::ReadBufferAndDumpToStream<StringPrinter::StringElementType::ASCII> (const ReadBufferAndDumpToStreamOptions& options)
630 // treat ASCII the same as UTF8
631 // FIXME: can we optimize ASCII some more?
632 return ReadBufferAndDumpToStream<StringElementType::UTF8>(options);
637 StringPrinter::ReadBufferAndDumpToStream<StringPrinter::StringElementType::UTF16> (const ReadBufferAndDumpToStreamOptions& options)
639 assert(options.GetStream() && "need a Stream to print the string to");
641 return DumpUTFBufferToStream(ConvertUTF16toUTF8, options);
646 StringPrinter::ReadBufferAndDumpToStream<StringPrinter::StringElementType::UTF32> (const ReadBufferAndDumpToStreamOptions& options)
648 assert(options.GetStream() && "need a Stream to print the string to");
650 return DumpUTFBufferToStream(ConvertUTF32toUTF8, options);
653 } // namespace formatters
655 } // namespace lldb_private