Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -24,6 +24,10 @@ and this project adheres to [Semantic Versioning](http://semver.org/spec/v2.0.0.
- Python 3.14 is now tested and declared as supported. No source changes were needed; the full
suite passes on 3.14 as-is.

### Fixed

- `heredocs_to_strings` writes the heredoc's value rather than its own text. It was quoting the source -- markers and all -- so `<<EOT\nhello\nEOT` became `"<<EOT\nhello\nEOT"`, a quoted string spanning three physical lines. A quoted template cannot span lines, so OpenTofu rejects that with "Invalid multi-line string", and reading it back here gave the marker text rather than the value: neither a valid file nor the right content. It now reuses the flattening the reader already performs, so the two cannot drift. A heredoc whose `${...}` runs across lines has no quoted spelling -- the newlines inside the expression can be neither escaped nor left raw -- so it stays a heredoc; flattening one raised `UnexpectedToken` out of `dumps`. So does one using a `~` strip marker, which strips across the line break in a quoted string but not in a heredoc, so the flattened form would be a different value. ([#337](https://github.com/amplify-education/python-hcl2/issues/337))

## \[8.1.4\] - 2026-09-08

### Fixed
Expand Down
109 changes: 102 additions & 7 deletions hcl2/deserializer.py
Original file line number Diff line number Diff line change
Expand Up @@ -61,7 +61,80 @@
IntLiteral,
)
from hcl2.transformer import RuleTransformer
from hcl2.utils import HEREDOC_PATTERN, HEREDOC_TRIM_PATTERN
from hcl2.utils import HEREDOC_PATTERN, HEREDOC_TRIM_PATTERN, SerializationOptions


def _has_quoted_spelling(heredoc: str) -> bool:
"""Whether *heredoc* can be written as a quoted string of the same value.

Two things rule it out: a `${...}` or `%{...}` that runs across a line,
and a `~` strip marker. A heredoc is lexed a line at a time, so `~}` there
strips only to the end of its own line, where in a quoted string it strips
across the newline and the next line's indent -- OpenTofu v1.12.6 gives the
two spellings of the same loop different values. Such a heredoc stays one.
"""
if "${~" in heredoc or "%{~" in heredoc or "~}" in heredoc:
return False
return not _has_multi_line_interpolation(heredoc)


def _has_multi_line_interpolation(heredoc: str) -> bool:
"""Whether a `${...}` or `%{...}` in *heredoc* runs across a line.

Such a heredoc has no quoted spelling: the newlines inside the span are
expression source, which OpenTofu rejects escaped ("This character is not
used within the language") and rejects raw, since a quoted string cannot
span lines. `heredocs_to_strings` leaves it a heredoc rather than raising.

Braces that are not structural are skipped: those in a string literal and
those in a comment. OpenTofu evaluates `${1 /* } */\n+ 2}` to 3, so the
span runs on past that brace and across the line. A `#` or `//` comment
runs to the end of its line, so one inside a span means the span crosses a
line. A string literal that itself opens `${` or `%{` is answered as
multi-line without looking further: keeping the heredoc is always safe,
and flattening it wrongly is not.
"""
index = 0
length = len(heredoc)
while index < length:
if heredoc.startswith(("$${", "%%{"), index):
index += 3
continue
if not heredoc.startswith(("${", "%{"), index):
index += 1
continue
depth = 0
in_string = False
index += 1
while index < length:
char = heredoc[index]
if char == "\n":
return True
if in_string:
if char == "\\":
index += 1
elif char == '"':
in_string = False
elif heredoc.startswith(("${", "%{"), index):
return True
elif char == '"':
in_string = True
elif char == "#" or heredoc.startswith("//", index):
return True
elif heredoc.startswith("/*", index):
end = heredoc.find("*/", index + 2)
if end == -1 or "\n" in heredoc[index:end]:
return True
index = end + 1
elif char == "{":
depth += 1
elif char == "}":
depth -= 1
if depth == 0:
break
index += 1
index += 1
return False


@dataclass
Expand Down Expand Up @@ -168,15 +241,19 @@ def _deserialize_text(self, value: Any) -> LarkRule:

if isinstance(value, str):
if value.startswith('"') and value.endswith('"'):
if not self.options.heredocs_to_strings and value.startswith('"<<-'):
match = HEREDOC_TRIM_PATTERN.match(value[1:-1])
if match:
if value.startswith('"<<-') and HEREDOC_TRIM_PATTERN.match(value[1:-1]):
if not self.options.heredocs_to_strings:
return self._deserialize_heredoc(value[1:-1], True)
if not _has_quoted_spelling(value[1:-1]):
return self._deserialize_heredoc(value[1:-1], True)
return self._deserialize_string(self._heredoc_as_quoted(value[1:-1], True))

if not self.options.heredocs_to_strings and value.startswith('"<<'):
match = HEREDOC_PATTERN.match(value[1:-1])
if match:
if value.startswith('"<<') and HEREDOC_PATTERN.match(value[1:-1]):
if not self.options.heredocs_to_strings:
return self._deserialize_heredoc(value[1:-1], False)
if not _has_quoted_spelling(value[1:-1]):
return self._deserialize_heredoc(value[1:-1], False)
return self._deserialize_string(self._heredoc_as_quoted(value[1:-1], False))

if self.options.strings_to_heredocs:
inner = value[1:-1]
Expand Down Expand Up @@ -252,6 +329,24 @@ def _deserialize_string_part(self, value: str) -> StringPartRule:

return StringPartRule([STRING_CHARS(value)])

def _heredoc_as_quoted(self, heredoc: str, trim: bool) -> str:
"""Return the quoted-string source for *heredoc*'s value.

Not by quoting the heredoc's own text: that is what this option used to
do, and it produced `"<<EOT\nhello\nEOT"` -- a quoted string spanning
three physical lines, markers and all. A quoted template cannot span
lines, so OpenTofu rejects it with "Invalid multi-line string", and
reading it back here gave the marker text rather than the value.

The flattening the reader already performs is reused rather than
written a second time, so the two cannot drift: serializing the rule
with `preserve_heredocs=False` is exactly the quoted form
`preserve_heredocs=False` produces on the way in.
"""
rule = self._deserialize_heredoc(heredoc, trim)
quoted: str = rule.serialize(SerializationOptions(preserve_heredocs=False))
return quoted

def _deserialize_heredoc(
self, value: str, trim: bool
) -> Union[HeredocTemplateRule, HeredocTrimTemplateRule]:
Expand Down
149 changes: 149 additions & 0 deletions test/unit/test_heredocs_to_strings.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,149 @@
# pylint: disable=C0103,C0114,C0115,C0116
r"""`heredocs_to_strings` writes a value, not the heredoc's own text (GH #337).

The option converts a heredoc into a quoted string. It was quoting the
heredoc's *source* -- markers and all -- across as many physical lines as the
original occupied:

a = "<<EOT
hello
EOT"

A quoted template cannot span lines, so OpenTofu rejects that with "Invalid
multi-line string", and reading it back here gave the marker text rather than
the value. Neither a valid file nor the right content.

The flattening the reader already performs is reused rather than written a
second time, so the two cannot drift.
"""

from unittest import TestCase

from hcl2.api import dumps, loads
from hcl2.deserializer import DeserializerOptions

STRINGS = DeserializerOptions(heredocs_to_strings=True)


class TestTheOutputIsAQuotedValue(TestCase):
def _convert(self, source: str) -> str:
return dumps(loads(source), deserializer_options=STRINGS)

def test_a_plain_heredoc(self):
self.assertEqual(self._convert("a = <<EOT\nhello\nEOT\n"), 'a = "hello"\n')

def test_a_trimmed_heredoc(self):
self.assertEqual(self._convert("a = <<-EOT\n indented\n EOT\n"), 'a = "indented"\n')

def test_quotes_in_the_body_are_escaped(self):
self.assertEqual(self._convert('a = <<EOT\nsay "hi"\nEOT\n'), 'a = "say \\"hi\\""\n')

def test_the_result_is_one_line(self):
for source in ("a = <<EOT\nhello\nEOT\n", "a = <<EOT\none\ntwo\nEOT\n"):
with self.subTest(source=source):
written = self._convert(source)
self.assertEqual(written.count("\n"), 1, written)

def test_the_result_parses_again(self):
for source in (
"a = <<EOT\nhello\nEOT\n",
'a = <<EOT\nsay "hi"\nEOT\n',
"a = <<EOT\none\ntwo\nEOT\n",
"a = <<-EOT\n indented\n EOT\n",
):
with self.subTest(source=source):
loads(self._convert(source))

def test_no_marker_survives_into_the_output(self):
written = self._convert("a = <<EOT\nhello\nEOT\n")
self.assertNotIn("EOT", written)
self.assertNotIn("<<", written)


class TestTheOptionOffIsUnchanged(TestCase):
def test_a_heredoc_stays_a_heredoc(self):
source = "a = <<EOT\nhello\nEOT\n"
self.assertEqual(dumps(loads(source)), source)

def test_a_plain_string_is_unaffected_either_way(self):
source = 'a = "hello"\n'
self.assertEqual(dumps(loads(source), deserializer_options=STRINGS), source)
self.assertEqual(dumps(loads(source)), source)


class TestAMultiLineInterpolationKeepsTheHeredoc(TestCase):
"""A heredoc whose `${...}` spans lines has no quoted spelling.

The newlines inside the span are expression source: escaped, OpenTofu
rejects them as "not used within the language", and raw, they make the
quoted string span lines. Flattening one raised `UnexpectedToken` out of
`dumps`. It is written back as the heredoc it was, which is the only answer
that keeps the file readable and the value unchanged.
"""

OPTIONS = DeserializerOptions(heredocs_to_strings=True)

def test_both_forms_are_kept(self):
for source in (
"x = <<EOF\na ${\n b\n} c\nEOF\n",
"x = <<-EOF\n a ${\n b\n } c\n EOF\n",
"x = <<EOF\n%{ if\n true }y%{ endif }\nEOF\n",
):
with self.subTest(source=source):
self.assertEqual(dumps(loads(source), deserializer_options=self.OPTIONS), source)

def test_a_one_line_interpolation_is_still_flattened(self):
written = dumps(loads("x = <<EOF\na ${b} c\nEOF\n"), deserializer_options=self.OPTIONS)
self.assertTrue(written.startswith('x = "a ${b} c'), written)

def test_a_brace_in_a_string_inside_the_span_does_not_end_it(self):
source = 'x = <<EOF\na ${f("}",\n b)} c\nEOF\n'
self.assertEqual(dumps(loads(source), deserializer_options=self.OPTIONS), source)


class TestACommentInsideTheSpanDoesNotEndIt(TestCase):
"""A brace inside a comment in `${...}` is not structural.

OpenTofu v1.12.6 evaluates each of these to `a 3 b\\n`: the `}` in the
comment does not close the expression, so the newline after it is still
expression source and the heredoc has no quoted spelling.
"""

OPTIONS = DeserializerOptions(heredocs_to_strings=True)

def test_each_comment_form(self):
for source in (
"x = <<EOF\na ${1 /* } */\n+ 2} b\nEOF\n",
"x = <<EOF\na ${1 # }\n+ 2} b\nEOF\n",
"x = <<EOF\na ${1 // }\n+ 2} b\nEOF\n",
"x = <<-EOF\n a ${1 /* } */\n + 2} b\n EOF\n",
):
with self.subTest(source=source):
self.assertEqual(dumps(loads(source), deserializer_options=self.OPTIONS), source)


class TestAStripMarkerKeepsTheHeredoc(TestCase):
"""`~` strips whitespace up to the line's end in a heredoc, and further in a string.

A heredoc body is lexed a line at a time, so a `~}` there strips only to
the end of its own line; in a quoted string the same marker strips across
the newline and the next line's indent. OpenTofu v1.12.6 evaluates the
loop below to `items:\\n - a\\n - b\\n`, and its flattened form to the
same list without the indent. No quoted spelling keeps the value, so the
heredoc stays one.
"""

OPTIONS = DeserializerOptions(heredocs_to_strings=True)

def test_each_marker_keeps_the_heredoc(self):
for source in (
'x = <<EOF\nitems:\n%{ for s in ["a", "b"] ~}\n - ${s}\n%{ endfor ~}\nEOF\n',
"x = <<EOF\na\n${~ b}\nEOF\n",
"x = <<EOF\na\n%{~ if true }y%{ endif }\nEOF\n",
):
with self.subTest(source=source):
self.assertEqual(dumps(loads(source), deserializer_options=self.OPTIONS), source)

def test_a_tilde_in_the_text_is_not_a_marker(self):
written = dumps(loads("x = <<EOF\na ~ b ${c}\nEOF\n"), deserializer_options=self.OPTIONS)
self.assertTrue(written.startswith('x = "a ~ b ${c}'), written)