Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
20 changes: 20 additions & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -50,6 +50,26 @@ dfp.baseimage = 'centos:7'
print(dfp.content)
```

### Here-documents

The parser groups [Dockerfile here-documents] with their `RUN`, `COPY`, or
`ADD` instruction, including instructions nested in `ONBUILD`. Quoted
delimiters, `<<-` tab stripping, and multiple bodies in declaration order
are supported. Body lines are not parsed as instructions or comments.

Each structure entry's `content` includes its complete heredoc bodies and
terminators, preserving their whitespace and line endings. Its `startline`
and `endline` cover the whole block, so inserting after, replacing, or
deleting the instruction also handles its bodies. The `value` and JSON
representation contain the normalized header followed by body and terminator
lines separated by `\n`, without a final newline. An unterminated heredoc
raises `ValueError` when reading `structure`, including edits that first
parse the instructions, such as setting `baseimage`. Cached and uncached
parsing use the same physical line boundaries; literal carriage returns
within a body do not create additional lines.

[Dockerfile here-documents]: https://docs.docker.com/reference/dockerfile/#here-documents

[coveralls status badge]: https://coveralls.io/repos/containerbuildsystem/dockerfile-parse/badge.svg?branch=master
[coveralls status link]: https://coveralls.io/r/containerbuildsystem/dockerfile-parse?branch=master
[lgtm python badge]: https://img.shields.io/lgtm/grade/python/g/containerbuildsystem/dockerfile-parse.svg?logo=lgtm&logoWidth=18
Expand Down
126 changes: 108 additions & 18 deletions dockerfile_parse/parser.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@
import os
import re
from contextlib import contextmanager
from io import StringIO
from shlex import quote

from .constants import DOCKERFILE_FILENAME, COMMENT_INSTRUCTION
Expand All @@ -22,6 +23,73 @@
logger = logging.getLogger(__name__)


def _dequote_heredoc(word):
"""Remove delimiter quotes without expanding variables or shell commands.

Docker's literal delimiter decoding uses these double-quote escape rules:
https://github.com/moby/buildkit/blob/v0.32.2/frontend/dockerfile/shell/lex.go#L315-L330
"""
result = []
quote_char = None
characters = iter(word)
for character in characters:
if character == '\\' and quote_char != "'":
escaped = next(characters, '')
# Docker's double-quote lexer only unescapes these three characters.
if quote_char == '"' and escaped not in ('"', '$', '\\'):
result.append('\\')
result.append(escaped)
elif character == quote_char:
quote_char = None
elif quote_char is None and character in ('"', "'"):
quote_char = character
else:
result.append(character)
if quote_char is not None:
raise ValueError('Unterminated quote in heredoc delimiter: {0}'.format(word))
return ''.join(result)


def _heredoc_delimiters(instruction, value):
"""Return (delimiter, strip_tabs) pairs in the header's declaration order.

Match Docker's whole-token recognition and whitespace after the operator:
https://github.com/moby/buildkit/blob/v0.32.2/frontend/dockerfile/parser/parser.go#L121
https://github.com/moby/buildkit/blob/v0.32.2/frontend/dockerfile/shell/lex.go#L516-L532
"""
if instruction == 'ONBUILD':
nested = value.split(None, 1)
if len(nested) != 2:
return []
instruction, value = nested
if instruction.upper() not in ('RUN', 'COPY', 'ADD') or '<<' not in value:
return []

delimiters = []
# Preserve quotes so a literal "<<EOF" cannot become a heredoc opener.
words = iter(WordSplitter(value).split(dequote=False, keep_whitespace=True))
for word in words:
# Docker keeps whitespace immediately after << in the same token.
# Whitespace after <<- instead ends the token, as for other words.
if re.fullmatch(r'[0-9]*<<', word):
separator = next(words, '')
if separator not in (' ', '\t', '\r'):
continue
while separator in (' ', '\t', '\r'):
word += separator
separator = next(words, '')
if separator.isspace():
continue
word += separator
match = re.fullmatch(r'[0-9]*<<(-?)[ \t\n\f\r]*([^<]*)', word)
if match:
chomp, quoted_delimiter = match.groups()
delimiter = _dequote_heredoc(quoted_delimiter)
if delimiter:
delimiters.append((delimiter, bool(chomp)))
return delimiters


class KeyValues(dict):
"""
Abstract base class for allowing direct write access to Dockerfile
Expand Down Expand Up @@ -158,7 +226,8 @@ def lines(self):
:return: list containing lines (unicode) from Dockerfile
"""
if self.cache_content and self.cached_content:
return self.cached_content.splitlines(True)
# Match binary readlines(): only LF separates physical lines.
return StringIO(self.cached_content).readlines()

try:
with self._open_dockerfile('rb') as dockerfile:
Expand Down Expand Up @@ -237,12 +306,16 @@ def structure(self):
"content": "CMD yum -y update && \\\n yum clean all\n",
"value": "yum -y update && yum clean all"}
]

Heredoc content and line bounds include the bodies and terminators.
Their value retains the normalized header followed by the literal body
and terminator lines, joined with newlines and without a final newline.
"""
def _rstrip_eol(text, line_continuation_char='\\'):
text = text.rstrip()
if text.endswith(line_continuation_char):
return text[:-1]
return text
def _continuation_regex(escape):
# Docker checks the immediately preceding character, not odd/even
# runs of escapes: even three trailing escapes end the instruction.
# https://github.com/moby/buildkit/blob/v0.32.2/frontend/dockerfile/parser/parser.go#L158-L168
return re.compile(r'(?<!{0}){0}[ \t]*$'.format(re.escape(escape)))

def _create_instruction_dict(instruction=None, value=None):
return {
Expand All @@ -259,10 +332,8 @@ def _clean_comment_line(line):
return line

instructions = []
lineno = -1
line_continuation_char = '\\'
insnre = re.compile(r'^\s*(\S+)\s+(.*)$') # matched group is insn
contre = re.compile(r'^.*\\\s*$') # line continues?
contre = _continuation_regex('\\') # line continues?
commentre = re.compile(r'^\s*#') # line is a comment?
directive_possible = True
# escape directive regex
Expand All @@ -273,17 +344,16 @@ def _clean_comment_line(line):
in_continuation = False
current_instruction = {}

for line in self.lines:
lineno += 1
lines = iter(enumerate(self.lines))
for lineno, line in lines:

if directive_possible:
# once support for python versions before 3.8 is dropped use walrus operator
if escape_directive_re.match(line):
# Do the matching twice if there is a directive to avoid doing the matching
# for other lines
match = escape_directive_re.match(line)
line_continuation_char = match.group(1)
contre = re.compile(r'^.*' + re.escape(match.group(1)) + r'\s*$')
contre = _continuation_regex(match.group(1))
elif syntax_directive_re.match(line):
# Currently no information for the syntax directive is stored it is still
# necessary to detect escape directives after a syntax directive
Expand All @@ -301,26 +371,46 @@ def _clean_comment_line(line):
instructions.append(comment)

else:
value = line.rstrip('\r\n')
continuation = contre.search(value)
value = value[:continuation.start()] if continuation else value.rstrip()
if not in_continuation:
m = insnre.match(line)
if not m:
continue
current_instruction = _create_instruction_dict(
instruction=m.groups()[0].upper(),
value=_rstrip_eol(m.groups()[1], line_continuation_char)
value=value[m.start(2):]
)
else:
current_instruction['content'] += line
current_instruction['endline'] = lineno

if current_instruction['value']:
current_instruction['value'] += _rstrip_eol(line, line_continuation_char)
current_instruction['value'] += value
else:
current_instruction['value'] = _rstrip_eol(line.lstrip(),
line_continuation_char)
current_instruction['value'] = value.lstrip()

in_continuation = contre.match(line)
in_continuation = continuation is not None
if not in_continuation and current_instruction:
heredocs = _heredoc_delimiters(current_instruction['instruction'],
current_instruction['value'])
for delimiter, strip_tabs in heredocs:
# Bodies are opaque: comments and continuation characters
# have no instruction-level meaning until the terminator.
for lineno, line in lines:
current_instruction['content'] += line
current_instruction['endline'] = lineno
body_line = line.rstrip('\r\n')
current_instruction['value'] += '\n' + body_line
if strip_tabs:
body_line = body_line.lstrip('\t')
if body_line == delimiter:
break
else:
raise ValueError(
'Unterminated heredoc {0!r} starting at line {1}'.format(
delimiter, current_instruction['startline'] + 1))
instructions.append(current_instruction)

return instructions
Expand Down
7 changes: 5 additions & 2 deletions dockerfile_parse/util.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,7 +33,7 @@ class WordSplitter(object):
dequote()
Returns the string with escaped and quotes consumed

split(maxsplit=None, dequote=True)
split(maxsplit=None, dequote=True, keep_whitespace=False)
Returns an iterable of words, split at whitespace
"""

Expand Down Expand Up @@ -100,13 +100,14 @@ def _update_quoting_state(self, ch):
def dequote(self):
return ''.join(self.split(maxsplit=0))

def split(self, maxsplit=None, dequote=True):
def split(self, maxsplit=None, dequote=True, keep_whitespace=False):
"""
Generator for the words of the string

:param maxsplit: perform at most maxsplit splits;
if None, do not limit the number of splits
:param dequote: remove quotes and escape characters once consumed
:param keep_whitespace: yield each separator character as a separate item
"""

class Word(object):
Expand Down Expand Up @@ -202,6 +203,8 @@ def append(self, s):
num_splits += 1
yield word.value

if keep_whitespace:
yield ch
word = Word()
else:
word.append(ch)
Expand Down
54 changes: 54 additions & 0 deletions tests/test_continuations.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,54 @@
"""Instruction boundaries around Dockerfile escape characters."""

import pytest

from tests.fixtures import dfparser

dfparser = dfparser # pylint: disable=self-assigning-variable


@pytest.mark.parametrize('escape', ['\\', '`'])
@pytest.mark.parametrize('count', [1, 2, 3])
@pytest.mark.parametrize('suffix', ['', ' \t', '\u00a0'])
@pytest.mark.parametrize('newline', ['\n', '\r\n'])
def test_terminal_escape_instruction_boundaries(dfparser, escape, count, suffix, newline):
directive = '# escape={}'.format(escape) + newline
header = 'RUN echo ' + escape * count + suffix + newline
following = 'FROM next' + newline
dfparser.content = directive + header + following
continues = count == 1 and suffix != '\u00a0'
expected = [
{'instruction': 'RUN', 'startline': 1, 'endline': 2 if continues else 1,
'content': header + following if continues else header,
'value': 'echo FROM next' if continues else 'echo ' + escape * count},
]
if not continues:
expected.append({'instruction': 'FROM', 'startline': 2, 'endline': 2,
'content': following, 'value': 'next'})
assert dfparser.structure[1:] == expected
assert dfparser.parent_images == ([] if continues else ['next'])


@pytest.mark.parametrize('escape', ['\\', '`'])
def test_escaped_escape_ends_continued_instruction(dfparser, escape):
header = 'RUN echo {0}\n value {0}{0}\n'.format(escape)
dfparser.content = '# escape={}\n'.format(escape) + header + 'FROM next\n'
assert dfparser.structure[1:] == [
{'instruction': 'RUN', 'startline': 1, 'endline': 2,
'content': header, 'value': 'echo value ' + escape * 2},
{'instruction': 'FROM', 'startline': 3, 'endline': 3,
'content': 'FROM next\n', 'value': 'next'},
]


@pytest.mark.parametrize('escape', ['\\', '`'])
def test_heredoc_header_ends_at_escaped_escape(dfparser, escape):
header = 'RUN cat <<EOF {0}{0}\n'.format(escape)
body = 'FROM payload\nEOF\n'
dfparser.content = '# escape={}\n'.format(escape) + header + body + 'FROM next\n'
assert dfparser.structure[1:] == [
{'instruction': 'RUN', 'startline': 1, 'endline': 3,
'content': header + body, 'value': header[4:-1] + '\n' + body.rstrip('\n')},
{'instruction': 'FROM', 'startline': 4, 'endline': 4,
'content': 'FROM next\n', 'value': 'next'},
]
Loading