From e2ab44df32b8c5aee10d24b7464decd597335f80 Mon Sep 17 00:00:00 2001 From: Tim Date: Tue, 1 Sep 2026 23:04:19 -0700 Subject: [PATCH 1/4] fix: heredocs_to_strings writes a value, not the heredoc's text (#337) The option converts a heredoc into a quoted string, and was quoting the heredoc's own source -- markers and all -- across as many physical lines as the original occupied: a = "< LarkRule: if isinstance(value, str): if value.startswith('"') and value.endswith('"'): - if not self.options.heredocs_to_strings and value.startswith('"<<-'): - match = HEREDOC_TRIM_PATTERN.match(value[1:-1]) - if match: + if value.startswith('"<<-') and HEREDOC_TRIM_PATTERN.match(value[1:-1]): + if not self.options.heredocs_to_strings: return self._deserialize_heredoc(value[1:-1], True) + return self._deserialize_string(self._heredoc_as_quoted(value[1:-1], True)) - if not self.options.heredocs_to_strings and value.startswith('"<<'): - match = HEREDOC_PATTERN.match(value[1:-1]) - if match: + if value.startswith('"<<') and HEREDOC_PATTERN.match(value[1:-1]): + if not self.options.heredocs_to_strings: return self._deserialize_heredoc(value[1:-1], False) + return self._deserialize_string(self._heredoc_as_quoted(value[1:-1], False)) if self.options.strings_to_heredocs: inner = value[1:-1] @@ -252,6 +252,24 @@ def _deserialize_string_part(self, value: str) -> StringPartRule: return StringPartRule([STRING_CHARS(value)]) + def _heredoc_as_quoted(self, heredoc: str, trim: bool) -> str: + """Return the quoted-string source for *heredoc*'s value. + + Not by quoting the heredoc's own text: that is what this option used to + do, and it produced `"< Union[HeredocTemplateRule, HeredocTrimTemplateRule]: diff --git a/test/unit/test_heredocs_to_strings.py b/test/unit/test_heredocs_to_strings.py new file mode 100644 index 00000000..0b7ce646 --- /dev/null +++ b/test/unit/test_heredocs_to_strings.py @@ -0,0 +1,71 @@ +# pylint: disable=C0103,C0114,C0115,C0116 +r"""`heredocs_to_strings` writes a value, not the heredoc's own text (GH #337). + +The option converts a heredoc into a quoted string. It was quoting the +heredoc's *source* -- markers and all -- across as many physical lines as the +original occupied: + + a = "< str: + return dumps(loads(source), deserializer_options=STRINGS) + + def test_a_plain_heredoc(self): + self.assertEqual(self._convert("a = < Date: Wed, 23 Sep 2026 12:01:27 -0700 Subject: [PATCH 2/4] fix: heredocs_to_strings keeps a heredoc it cannot flatten A heredoc whose `${...}` runs across lines has no quoted spelling: the newlines inside the span are expression source, which OpenTofu rejects escaped and rejects raw. Flattening one raised UnexpectedToken out of dumps. It is now written back as the heredoc it was. --- CHANGELOG.md | 2 +- hcl2/deserializer.py | 49 +++++++++++++++++++++++++++ test/unit/test_heredocs_to_strings.py | 30 ++++++++++++++++ 3 files changed, 80 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2e18d7b8..f294569c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -26,7 +26,7 @@ and this project adheres to [Semantic Versioning](http://semver.org/spec/v2.0.0. ### Fixed -- `heredocs_to_strings` writes the heredoc's value rather than its own text. It was quoting the source -- markers and all -- so `< bool: + """Whether a `${...}` or `%{...}` in *heredoc* runs across a line. + + Such a heredoc has no quoted spelling: the newlines inside the span are + expression source, which OpenTofu rejects escaped ("This character is not + used within the language") and rejects raw, since a quoted string cannot + span lines. `heredocs_to_strings` leaves it a heredoc rather than raising. + + A `"` inside the span opens a string literal whose braces do not count; + the escapes `$${` and `%%{` open nothing. + """ + index = 0 + length = len(heredoc) + while index < length: + if heredoc.startswith(("$${", "%%{"), index): + index += 3 + continue + if not heredoc.startswith(("${", "%{"), index): + index += 1 + continue + depth = 0 + in_string = False + index += 1 + while index < length: + char = heredoc[index] + if in_string: + if char == "\\": + index += 1 + elif char == '"': + in_string = False + elif char == '"': + in_string = True + elif char == "{": + depth += 1 + elif char == "}": + depth -= 1 + if depth == 0: + break + if char == "\n": + return True + index += 1 + index += 1 + return False + + @dataclass class DeserializerOptions: """Options controlling how Python dicts are deserialized into LarkElement trees.""" @@ -171,11 +216,15 @@ def _deserialize_text(self, value: Any) -> LarkRule: if value.startswith('"<<-') and HEREDOC_TRIM_PATTERN.match(value[1:-1]): if not self.options.heredocs_to_strings: return self._deserialize_heredoc(value[1:-1], True) + if _has_multi_line_interpolation(value[1:-1]): + return self._deserialize_heredoc(value[1:-1], True) return self._deserialize_string(self._heredoc_as_quoted(value[1:-1], True)) if value.startswith('"<<') and HEREDOC_PATTERN.match(value[1:-1]): if not self.options.heredocs_to_strings: return self._deserialize_heredoc(value[1:-1], False) + if _has_multi_line_interpolation(value[1:-1]): + return self._deserialize_heredoc(value[1:-1], False) return self._deserialize_string(self._heredoc_as_quoted(value[1:-1], False)) if self.options.strings_to_heredocs: diff --git a/test/unit/test_heredocs_to_strings.py b/test/unit/test_heredocs_to_strings.py index 0b7ce646..1d612415 100644 --- a/test/unit/test_heredocs_to_strings.py +++ b/test/unit/test_heredocs_to_strings.py @@ -69,3 +69,33 @@ def test_a_plain_string_is_unaffected_either_way(self): source = 'a = "hello"\n' self.assertEqual(dumps(loads(source), deserializer_options=STRINGS), source) self.assertEqual(dumps(loads(source)), source) + + +class TestAMultiLineInterpolationKeepsTheHeredoc(TestCase): + """A heredoc whose `${...}` spans lines has no quoted spelling. + + The newlines inside the span are expression source: escaped, OpenTofu + rejects them as "not used within the language", and raw, they make the + quoted string span lines. Flattening one raised `UnexpectedToken` out of + `dumps`. It is written back as the heredoc it was, which is the only answer + that keeps the file readable and the value unchanged. + """ + + OPTIONS = DeserializerOptions(heredocs_to_strings=True) + + def test_both_forms_are_kept(self): + for source in ( + "x = < Date: Wed, 23 Sep 2026 12:40:43 -0700 Subject: [PATCH 3/4] fix: a comment inside the span does not end it The multi-line check counted a brace inside a comment as closing the interpolation, so `${1 /* } */\n+ 2}` -- which OpenTofu evaluates to 3 -- was judged single-line, flattened, and dumps raised UnexpectedToken. Block comments are skipped, a line comment inside a span means the span crosses a line, and a string literal that opens a nested template keeps the heredoc, which is always safe. --- hcl2/deserializer.py | 22 ++++++++++++++++++---- test/unit/test_heredocs_to_strings.py | 21 +++++++++++++++++++++ 2 files changed, 39 insertions(+), 4 deletions(-) diff --git a/hcl2/deserializer.py b/hcl2/deserializer.py index e788ff38..97f58bac 100644 --- a/hcl2/deserializer.py +++ b/hcl2/deserializer.py @@ -72,8 +72,13 @@ def _has_multi_line_interpolation(heredoc: str) -> bool: used within the language") and rejects raw, since a quoted string cannot span lines. `heredocs_to_strings` leaves it a heredoc rather than raising. - A `"` inside the span opens a string literal whose braces do not count; - the escapes `$${` and `%%{` open nothing. + Braces that are not structural are skipped: those in a string literal and + those in a comment. OpenTofu evaluates `${1 /* } */\n+ 2}` to 3, so the + span runs on past that brace and across the line. A `#` or `//` comment + runs to the end of its line, so one inside a span means the span crosses a + line. A string literal that itself opens `${` or `%{` is answered as + multi-line without looking further: keeping the heredoc is always safe, + and flattening it wrongly is not. """ index = 0 length = len(heredoc) @@ -89,21 +94,30 @@ def _has_multi_line_interpolation(heredoc: str) -> bool: index += 1 while index < length: char = heredoc[index] + if char == "\n": + return True if in_string: if char == "\\": index += 1 elif char == '"': in_string = False + elif heredoc.startswith(("${", "%{"), index): + return True elif char == '"': in_string = True + elif char == "#" or heredoc.startswith("//", index): + return True + elif heredoc.startswith("/*", index): + end = heredoc.find("*/", index + 2) + if end == -1 or "\n" in heredoc[index:end]: + return True + index = end + 1 elif char == "{": depth += 1 elif char == "}": depth -= 1 if depth == 0: break - if char == "\n": - return True index += 1 index += 1 return False diff --git a/test/unit/test_heredocs_to_strings.py b/test/unit/test_heredocs_to_strings.py index 1d612415..71617ebc 100644 --- a/test/unit/test_heredocs_to_strings.py +++ b/test/unit/test_heredocs_to_strings.py @@ -99,3 +99,24 @@ def test_a_one_line_interpolation_is_still_flattened(self): def test_a_brace_in_a_string_inside_the_span_does_not_end_it(self): source = 'x = < Date: Wed, 23 Sep 2026 13:03:24 -0700 Subject: [PATCH 4/4] fix: heredocs_to_strings keeps a heredoc that uses a strip marker A heredoc is lexed a line at a time, so a ~} there strips only to the end of its line; in a quoted string it strips across the newline and the next line's indent. OpenTofu gives the two spellings of the same for-loop different values, so such a heredoc has no quoted spelling and stays one. --- CHANGELOG.md | 2 +- hcl2/deserializer.py | 18 ++++++++++++++++-- test/unit/test_heredocs_to_strings.py | 27 +++++++++++++++++++++++++++ 3 files changed, 44 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f294569c..6492e592 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -26,7 +26,7 @@ and this project adheres to [Semantic Versioning](http://semver.org/spec/v2.0.0. ### Fixed -- `heredocs_to_strings` writes the heredoc's value rather than its own text. It was quoting the source -- markers and all -- so `< bool: + """Whether *heredoc* can be written as a quoted string of the same value. + + Two things rule it out: a `${...}` or `%{...}` that runs across a line, + and a `~` strip marker. A heredoc is lexed a line at a time, so `~}` there + strips only to the end of its own line, where in a quoted string it strips + across the newline and the next line's indent -- OpenTofu v1.12.6 gives the + two spellings of the same loop different values. Such a heredoc stays one. + """ + if "${~" in heredoc or "%{~" in heredoc or "~}" in heredoc: + return False + return not _has_multi_line_interpolation(heredoc) + + def _has_multi_line_interpolation(heredoc: str) -> bool: """Whether a `${...}` or `%{...}` in *heredoc* runs across a line. @@ -230,14 +244,14 @@ def _deserialize_text(self, value: Any) -> LarkRule: if value.startswith('"<<-') and HEREDOC_TRIM_PATTERN.match(value[1:-1]): if not self.options.heredocs_to_strings: return self._deserialize_heredoc(value[1:-1], True) - if _has_multi_line_interpolation(value[1:-1]): + if not _has_quoted_spelling(value[1:-1]): return self._deserialize_heredoc(value[1:-1], True) return self._deserialize_string(self._heredoc_as_quoted(value[1:-1], True)) if value.startswith('"<<') and HEREDOC_PATTERN.match(value[1:-1]): if not self.options.heredocs_to_strings: return self._deserialize_heredoc(value[1:-1], False) - if _has_multi_line_interpolation(value[1:-1]): + if not _has_quoted_spelling(value[1:-1]): return self._deserialize_heredoc(value[1:-1], False) return self._deserialize_string(self._heredoc_as_quoted(value[1:-1], False)) diff --git a/test/unit/test_heredocs_to_strings.py b/test/unit/test_heredocs_to_strings.py index 71617ebc..08ff8431 100644 --- a/test/unit/test_heredocs_to_strings.py +++ b/test/unit/test_heredocs_to_strings.py @@ -120,3 +120,30 @@ def test_each_comment_form(self): ): with self.subTest(source=source): self.assertEqual(dumps(loads(source), deserializer_options=self.OPTIONS), source) + + +class TestAStripMarkerKeepsTheHeredoc(TestCase): + """`~` strips whitespace up to the line's end in a heredoc, and further in a string. + + A heredoc body is lexed a line at a time, so a `~}` there strips only to + the end of its own line; in a quoted string the same marker strips across + the newline and the next line's indent. OpenTofu v1.12.6 evaluates the + loop below to `items:\\n - a\\n - b\\n`, and its flattened form to the + same list without the indent. No quoted spelling keeps the value, so the + heredoc stays one. + """ + + OPTIONS = DeserializerOptions(heredocs_to_strings=True) + + def test_each_marker_keeps_the_heredoc(self): + for source in ( + 'x = <