mirror of https://github.com/interlegis/sapl.git
committed by
GitHub
57 changed files with 590 additions and 73 deletions
@ -0,0 +1,228 @@ |
|||
import pytest |
|||
from django.db.models.signals import pre_save |
|||
from django.template import Context, Template |
|||
from model_bakery import baker |
|||
|
|||
from sapl.compilacao.forms import TipoDispositivoForm |
|||
from sapl.compilacao.models import TipoDispositivo, TipoTextoArticulado |
|||
from sapl.crispy_layout_mixin import get_field_display |
|||
from sapl.lexml.models import LexmlProvedor |
|||
from sapl.parlamentares.models import Parlamentar |
|||
from sapl.protocoloadm.models import TramitacaoAdministrativo |
|||
from sapl.sanitize import (html_fragment_is_balanced, sanitize_field, |
|||
sanitize_html, sanitize_scope) |
|||
from sapl.sessao.models import ExpedienteSessao |
|||
|
|||
|
|||
def test_plain_remove_marcacao_e_preserva_texto(): |
|||
assert sanitize_html('<script>alert(1)</script>Ciente') == 'Ciente' |
|||
assert sanitize_html('Encaminhado <b>ao</b> setor') == \ |
|||
'Encaminhado ao setor' |
|||
assert sanitize_html('<img src=x onerror=alert(1)>') == '' |
|||
assert sanitize_html('<div>a</div><div>b</div>') == 'ab' |
|||
|
|||
|
|||
def test_plain_guarda_texto_puro_sem_escape(): |
|||
"""O escape é da renderização; gravado escapado, apareceria & na tela.""" |
|||
assert sanitize_html('Valor < 10 & prazo > 5') == 'Valor < 10 & prazo > 5' |
|||
assert sanitize_html('ALFA & BETA') == 'ALFA & BETA' |
|||
|
|||
|
|||
def test_plain_preserva_quebras_de_linha(): |
|||
# get_field_display converte \n em <br/> depois de sanitizar |
|||
assert sanitize_html('linha1\nlinha2') == 'linha1\nlinha2' |
|||
|
|||
|
|||
def test_valores_vazios_atravessam(): |
|||
assert sanitize_html('') == '' |
|||
assert sanitize_html(None) is None |
|||
assert sanitize_html('', rich=True) == '' |
|||
|
|||
|
|||
@pytest.mark.parametrize('valor', [ |
|||
'<script>alert(1)</script>Ciente', |
|||
'Valor < 10 & prazo > 5', |
|||
'Encaminhado <b>ao</b> <a href="https://x">setor</a>', |
|||
'texto & cia', |
|||
]) |
|||
def test_plain_e_idempotente(valor): |
|||
"""Salvar de novo um registro já sanitizado não altera o texto.""" |
|||
uma_vez = sanitize_html(valor) |
|||
assert sanitize_html(uma_vez) == uma_vez |
|||
|
|||
|
|||
@pytest.mark.parametrize('valor', [ |
|||
'<a href="https://camara.gov.br" target="_blank">Portal</a>', |
|||
'<p style="text-align: center;">centro</p>', |
|||
'<table><tr><td colspan="2">c</td></tr></table>', |
|||
'<script>alert(1)</script><b>ok</b>', |
|||
]) |
|||
def test_rich_e_idempotente(valor): |
|||
uma_vez = sanitize_html(valor, rich=True) |
|||
assert sanitize_html(uma_vez, rich=True) == uma_vez |
|||
|
|||
|
|||
def test_rich_preserva_links(): |
|||
saida = sanitize_html( |
|||
'<a href="https://camara.gov.br" target="_blank">Portal</a>', |
|||
rich=True) |
|||
assert 'href="https://camara.gov.br"' in saida |
|||
assert 'target="_blank"' in saida |
|||
assert 'rel="noopener noreferrer"' in saida |
|||
assert '>Portal</a>' in saida |
|||
|
|||
assert 'href="/materia/123"' in sanitize_html( |
|||
'<a href="/materia/123">Matéria</a>', rich=True) |
|||
assert 'href="mailto:a@b.c"' in sanitize_html( |
|||
'<a href="mailto:a@b.c">mail</a>', rich=True) |
|||
|
|||
|
|||
def test_rich_remove_href_perigosa_mas_mantem_o_texto(): |
|||
saida = sanitize_html( |
|||
'<a href="javascript:alert(1)">clique</a>', rich=True) |
|||
assert 'javascript' not in saida |
|||
assert 'clique' in saida |
|||
|
|||
|
|||
def test_rich_remove_script_e_manipuladores_de_evento(): |
|||
saida = sanitize_html('<script>alert(1)</script><b>ok</b>', rich=True) |
|||
assert saida == '<b>ok</b>' |
|||
|
|||
assert 'onclick' not in sanitize_html( |
|||
'<a href="#" onclick="steal()">x</a>', rich=True) |
|||
assert 'onerror' not in sanitize_html( |
|||
'<img src="x" onerror="alert(1)">', rich=True) |
|||
|
|||
|
|||
def test_rich_preserva_formatacao_do_tinymce(): |
|||
"""Protege contra regressão visual no conteúdo já cadastrado.""" |
|||
assert 'style="text-align: center;"' in sanitize_html( |
|||
'<p style="text-align: center;">centro</p>', rich=True) |
|||
|
|||
saida = sanitize_html( |
|||
'<table><tr><td colspan="2">c</td></tr></table>', rich=True) |
|||
assert '<table>' in saida and 'colspan="2"' in saida |
|||
|
|||
assert sanitize_html('<ul><li>a</li><li>b</li></ul>', rich=True) == \ |
|||
'<ul><li>a</li><li>b</li></ul>' |
|||
|
|||
|
|||
def test_sanitize_scope(): |
|||
assert sanitize_scope(TramitacaoAdministrativo, 'texto') == 'plain' |
|||
assert sanitize_scope(ExpedienteSessao, 'conteudo') == 'rich' |
|||
assert sanitize_scope(LexmlProvedor, 'xml') == 'exempt' |
|||
assert sanitize_scope(TipoTextoArticulado, 'rodape_global') == 'exempt' |
|||
assert sanitize_scope(Parlamentar, 'biografia') == 'rich' |
|||
|
|||
|
|||
def test_sanitize_field_respeita_isencao(): |
|||
xml = '<xml><a href="javascript:x">y</a></xml>' |
|||
assert sanitize_field(LexmlProvedor, 'xml', xml) == xml |
|||
|
|||
|
|||
@pytest.mark.django_db |
|||
def test_pre_save_sanitiza_campo_simples(): |
|||
t = baker.make(TramitacaoAdministrativo, |
|||
texto='<script>alert(1)</script>Ciente <b>ok</b>') |
|||
t.refresh_from_db() |
|||
assert t.texto == 'Ciente ok' |
|||
|
|||
|
|||
@pytest.mark.django_db |
|||
def test_pre_save_sanitiza_campo_rico_preservando_html(): |
|||
e = baker.make(ExpedienteSessao, |
|||
conteudo='<b>x</b><script>alert(1)</script>' |
|||
'<a href="https://a.b" target="_blank">l</a>') |
|||
e.refresh_from_db() |
|||
assert '<script>' not in e.conteudo |
|||
assert '<b>x</b>' in e.conteudo |
|||
assert 'href="https://a.b"' in e.conteudo |
|||
|
|||
|
|||
@pytest.mark.django_db |
|||
def test_pre_save_nao_toca_modelo_isento(): |
|||
xml = '<xml>a & b <script>x</script></xml>' |
|||
p = baker.make(LexmlProvedor, xml=xml) |
|||
p.refresh_from_db() |
|||
assert p.xml == xml |
|||
|
|||
|
|||
@pytest.mark.django_db |
|||
def test_get_field_display_nao_devolve_script(): |
|||
t = TramitacaoAdministrativo(texto='<script>alert(1)</script>Ciente') |
|||
__, display = get_field_display(t, 'texto') |
|||
assert '<script>' not in display |
|||
assert 'Ciente' in display |
|||
|
|||
|
|||
@pytest.mark.django_db |
|||
def test_get_field_display_protege_linha_legada(): |
|||
"""Linhas gravadas antes do pre_save não passam pela camada de entrada.""" |
|||
t = baker.make(TramitacaoAdministrativo, texto='ok') |
|||
TramitacaoAdministrativo.objects.filter(pk=t.pk).update( |
|||
texto='<script>alert(1)</script>legado') |
|||
t.refresh_from_db() |
|||
assert t.texto == '<script>alert(1)</script>legado' |
|||
|
|||
__, display = get_field_display(t, 'texto') |
|||
assert '<script>' not in display |
|||
|
|||
|
|||
def test_pre_save_ignora_raw(): |
|||
"""loaddata (inclusive em migrations) grava o objeto literalmente.""" |
|||
t = TramitacaoAdministrativo(texto='<b>fixture</b>') |
|||
pre_save.send(sender=TramitacaoAdministrativo, instance=t, raw=True) |
|||
assert t.texto == '<b>fixture</b>' |
|||
|
|||
|
|||
def test_get_field_display_escapa_texto_puro_uma_unica_vez(): |
|||
t = TramitacaoAdministrativo(texto='ALFA & BETA < 30') |
|||
__, display = get_field_display(t, 'texto') |
|||
assert 'ALFA & BETA < 30' in display |
|||
assert '&amp;' not in display |
|||
|
|||
|
|||
def test_get_field_display_escapa_campo_isento(): |
|||
p = LexmlProvedor(xml='<xml><script>x</script></xml>') |
|||
__, display = get_field_display(p, 'xml') |
|||
assert '<script>' not in display |
|||
assert '<xml>' in display |
|||
|
|||
|
|||
def test_striptags_apos_sanitize_nao_escapa_duas_vezes(): |
|||
"""Blocos da ata: texto puro, sem entidades escapadas de novo.""" |
|||
t = Template('{% load common_tags %}{{ v|sanitize|striptags }}') |
|||
saida = t.render(Context({ |
|||
'v': '<p>Ofício 12 lido & arquivado</p>' |
|||
'<script>alert(1)</script><img src=x onerror=alert(1)>'})) |
|||
assert saida == 'Ofício 12 lido & arquivado' |
|||
|
|||
|
|||
@pytest.mark.parametrize('valor, esperado', [ |
|||
('<br>', True), |
|||
('<br/>', True), |
|||
('<div class="titulo">Justificativa</div>', True), |
|||
('Art. ', True), |
|||
('<span class="x">', False), |
|||
('</span>', False), |
|||
('<b><i>x</b></i>', False), |
|||
]) |
|||
def test_html_fragment_is_balanced(valor, esperado): |
|||
assert html_fragment_is_balanced(valor) is esperado |
|||
|
|||
|
|||
@pytest.mark.django_db |
|||
def test_tipo_dispositivo_form_rejeita_fragmento_desbalanceado(): |
|||
td = baker.make(TipoDispositivo) |
|||
dados = {f: getattr(td, f) or '' for f in TipoDispositivoForm.Meta.fields} |
|||
dados['rotulo_prefixo_html'] = '<span class="rotulo">' |
|||
dados['rotulo_sufixo_html'] = '</span>' |
|||
form = TipoDispositivoForm(data=dados, instance=td) |
|||
assert not form.is_valid() |
|||
assert 'rotulo_prefixo_html' in form.errors |
|||
assert 'rotulo_sufixo_html' in form.errors |
|||
|
|||
dados['rotulo_prefixo_html'] = '<br/>' |
|||
dados['rotulo_sufixo_html'] = '' |
|||
form = TipoDispositivoForm(data=dados, instance=td) |
|||
assert 'rotulo_prefixo_html' not in form.errors |
|||
@ -0,0 +1,182 @@ |
|||
"""Sanitização de HTML/JavaScript nos campos de texto livre do SAPL. |
|||
|
|||
Módulo sem dependências internas do SAPL de propósito: é importado por |
|||
``sapl.crispy_layout_mixin``, ``sapl.base.receivers`` e pelos templatetags, |
|||
e ``sapl.utils`` já importa ``sapl.crispy_layout_mixin``. |
|||
|
|||
Duas políticas: |
|||
|
|||
* ``plain`` — remove toda a marcação e guarda texto puro, com as entidades |
|||
já decodificadas: o escape fica por conta da renderização. É o padrão para |
|||
qualquer ``TextField``. |
|||
* ``rich`` — allowlist para os campos editados via TinyMCE, que contêm HTML |
|||
legítimo (negrito, listas, tabelas e links). |
|||
|
|||
A ``rich`` é idempotente, o que permite aplicá-la tanto no ``pre_save`` quanto |
|||
na renderização. A ``plain`` só não é idempotente para texto que seja HTML |
|||
codificado em entidades (``<b>``): a primeira passada decodifica, a |
|||
segunda remove a tag. Na renderização, campos ``plain`` são escapados em vez |
|||
de re-sanitizados. |
|||
""" |
|||
|
|||
import html |
|||
from html.parser import HTMLParser |
|||
|
|||
import nh3 |
|||
|
|||
SANITIZE_RICH_TAGS = { |
|||
'p', 'br', 'hr', 'div', 'span', |
|||
'b', 'strong', 'i', 'em', 'u', 's', 'strike', 'sub', 'sup', |
|||
'ul', 'ol', 'li', 'dl', 'dt', 'dd', 'blockquote', 'pre', 'code', |
|||
'h1', 'h2', 'h3', 'h4', 'h5', 'h6', |
|||
'table', 'thead', 'tbody', 'tfoot', 'tr', 'th', 'td', 'caption', |
|||
'col', 'colgroup', |
|||
'a', 'img', |
|||
} |
|||
|
|||
# 'style' mantém o alinhamento produzido pelos botões do TinyMCE; |
|||
# 'target' mantém o "abrir em nova aba" dos links já cadastrados. |
|||
# 'rel' não pode entrar aqui: o nh3 aborta se a tag 'a' declarar 'rel' |
|||
# ao mesmo tempo em que link_rel está definido — ele mesmo escreve o atributo. |
|||
SANITIZE_RICH_ATTRS = { |
|||
'*': {'style', 'class', 'align', 'title', 'dir', 'lang'}, |
|||
'a': {'href', 'target', 'name'}, |
|||
'img': {'src', 'alt', 'width', 'height'}, |
|||
'td': {'colspan', 'rowspan', 'headers'}, |
|||
'th': {'colspan', 'rowspan', 'scope', 'headers'}, |
|||
'col': {'span'}, |
|||
'colgroup': {'span'}, |
|||
'table': {'border', 'cellpadding', 'cellspacing', 'summary'}, |
|||
} |
|||
|
|||
SANITIZE_URL_SCHEMES = {'http', 'https', 'mailto', 'tel'} |
|||
|
|||
# O conteúdo destas tags é descartado junto com a tag. Sem isso o texto de |
|||
# dentro de um <script> sobreviveria como texto solto. |
|||
SANITIZE_CLEAN_CONTENT_TAGS = {'script', 'style'} |
|||
|
|||
# Campos que guardam HTML legítimo, indexados por '<app_label>.<Model>'. |
|||
# Os de sessao e compilacao.Dispositivo são editados no TinyMCE; os de |
|||
# compilacao.TipoDispositivo são fragmentos de template configurados por |
|||
# administradores. |
|||
RICH_TEXT_FIELDS = { |
|||
'base.CasaLegislativa': {'informacao_geral'}, |
|||
'norma.NormaRelacionada': {'resumo'}, |
|||
'parlamentares.Legislatura': {'observacao'}, |
|||
'parlamentares.Parlamentar': {'biografia'}, |
|||
'sessao.ExpedienteSessao': {'conteudo'}, |
|||
'sessao.OcorrenciaSessao': {'conteudo'}, |
|||
'sessao.ConsideracoesFinais': {'conteudo'}, |
|||
'compilacao.Dispositivo': {'texto', 'texto_atualizador'}, |
|||
'compilacao.TipoDispositivo': { |
|||
'rotulo_prefixo_html', 'rotulo_sufixo_html', |
|||
'texto_prefixo_html', 'texto_sufixo_html', |
|||
'nota_automatica_prefixo_html', 'nota_automatica_sufixo_html', |
|||
}, |
|||
} |
|||
|
|||
# Campos que não devem ser tocados em hipótese alguma. |
|||
SANITIZE_EXEMPT_FIELDS = { |
|||
# xml é XML fornecido pela equipe do LexML; já é escapado em pretty_xml |
|||
'lexml.LexmlProvedor': {'xml'}, |
|||
# rodape_global é interpolado dentro de um content: de CSS |
|||
'compilacao.TipoTextoArticulado': {'rodape_global'}, |
|||
} |
|||
|
|||
|
|||
def model_key(model): |
|||
return '{}.{}'.format(model._meta.app_label, model.__name__) |
|||
|
|||
|
|||
def sanitize_scope(model, fieldname): |
|||
"""Retorna 'exempt', 'rich' ou 'plain' para um campo de um modelo.""" |
|||
key = model_key(model) |
|||
if fieldname in SANITIZE_EXEMPT_FIELDS.get(key, ()): |
|||
return 'exempt' |
|||
if fieldname in RICH_TEXT_FIELDS.get(key, ()): |
|||
return 'rich' |
|||
return 'plain' |
|||
|
|||
|
|||
def sanitize_html(value, rich=False): |
|||
"""Remove HTML/JavaScript perigoso de ``value``. |
|||
|
|||
Com ``rich=False`` toda a marcação é removida e apenas o texto sobra, |
|||
sem escape: ``&`` continua ``&``. |
|||
Com ``rich=True`` aplica-se a allowlist: links são preservados, mas |
|||
esquemas de URL fora de SANITIZE_URL_SCHEMES (javascript:, data:) e |
|||
manipuladores de evento (onclick, onerror) são descartados. |
|||
""" |
|||
if not value: |
|||
return value |
|||
|
|||
if not isinstance(value, str): |
|||
value = str(value) |
|||
|
|||
if rich: |
|||
return nh3.clean( |
|||
value, |
|||
tags=SANITIZE_RICH_TAGS, |
|||
attributes=SANITIZE_RICH_ATTRS, |
|||
url_schemes=SANITIZE_URL_SCHEMES, |
|||
clean_content_tags=SANITIZE_CLEAN_CONTENT_TAGS, |
|||
link_rel='noopener noreferrer', |
|||
strip_comments=True) |
|||
|
|||
return html.unescape(nh3.clean( |
|||
value, |
|||
tags=set(), |
|||
attributes={}, |
|||
clean_content_tags=SANITIZE_CLEAN_CONTENT_TAGS, |
|||
link_rel=None, |
|||
strip_comments=True)) |
|||
|
|||
|
|||
def sanitize_field(model, fieldname, value): |
|||
"""Sanitiza ``value`` conforme a política do campo. |
|||
|
|||
Campos isentos atravessam sem modificação. |
|||
""" |
|||
scope = sanitize_scope(model, fieldname) |
|||
if scope == 'exempt': |
|||
return value |
|||
return sanitize_html(value, rich=(scope == 'rich')) |
|||
|
|||
|
|||
_VOID_TAGS = { |
|||
'area', 'base', 'br', 'col', 'embed', 'hr', 'img', 'input', 'link', |
|||
'meta', 'param', 'source', 'track', 'wbr', |
|||
} |
|||
|
|||
|
|||
class _BalanceChecker(HTMLParser): |
|||
|
|||
def __init__(self): |
|||
super().__init__(convert_charrefs=True) |
|||
self.stack = [] |
|||
self.balanced = True |
|||
|
|||
def handle_starttag(self, tag, attrs): |
|||
if tag not in _VOID_TAGS: |
|||
self.stack.append(tag) |
|||
|
|||
def handle_startendtag(self, tag, attrs): |
|||
pass |
|||
|
|||
def handle_endtag(self, tag): |
|||
if tag in _VOID_TAGS: |
|||
return |
|||
if not self.stack or self.stack.pop() != tag: |
|||
self.balanced = False |
|||
|
|||
|
|||
def html_fragment_is_balanced(value): |
|||
"""Indica se toda tag aberta em ``value`` é fechada nele mesmo. |
|||
|
|||
Fragmentos desbalanceados (``<span>`` num campo, ``</span>`` em outro) |
|||
são reescritos pela sanitização ``rich``, que fecha ou descarta as tags. |
|||
""" |
|||
checker = _BalanceChecker() |
|||
checker.feed(value) |
|||
checker.close() |
|||
return checker.balanced and not checker.stack |
|||
@ -1,4 +1,5 @@ |
|||
{% load common_tags %} |
|||
<h2 class="gray-title">Considerações Finais</h2> |
|||
{% for c in lst_consideracoes%} |
|||
<p>{{c|striptags|safe}}</p> |
|||
<p>{{c|sanitize|striptags}}</p> |
|||
{% endfor %} |
|||
|
|||
@ -1,5 +1,6 @@ |
|||
{% load common_tags %} |
|||
<h2 class="gray-title">Expedientes</h2> |
|||
{% for expediente in lst_expedientes%} |
|||
<h3>{{expediente.nom_expediente}}</h3> |
|||
<div style="margin-bottom: 1cm">{{expediente.txt_expediente|safe}}</div> |
|||
<div style="margin-bottom: 1cm">{{expediente.txt_expediente|sanitize}}</div> |
|||
{% endfor%} |
|||
|
|||
@ -1,4 +1,5 @@ |
|||
{% load common_tags %} |
|||
<h2 class="gray-title">Ocorrências da Sessão</h2> |
|||
{% for o in lst_ocorrencias%} |
|||
<p>{{o.conteudo|striptags|safe}}</p> |
|||
<p>{{o.conteudo|sanitize|striptags}}</p> |
|||
{% endfor %} |
|||
|
|||
@ -1,8 +1,9 @@ |
|||
{% load common_tags %} |
|||
{% if object.consideracoesfinais.conteudo %} |
|||
<fieldset> |
|||
<p style="text-align: justify;margin-top: 0"> |
|||
<strong>Considerações Finais: </strong> |
|||
{{object.consideracoesfinais.conteudo|striptags|safe}} |
|||
{{object.consideracoesfinais.conteudo|sanitize|striptags}} |
|||
</p> |
|||
</fieldset> |
|||
{% endif %} |
|||
|
|||
@ -1,8 +1,9 @@ |
|||
{% load common_tags %} |
|||
{% if object.ocorrenciasessao.conteudo %} |
|||
<fieldset> |
|||
<p style="text-align: justify;margin-top: 0"> |
|||
<strong>Ocorrências da Sessão: </strong> |
|||
{{ object.ocorrenciasessao.conteudo|striptags|safe }} |
|||
{{ object.ocorrenciasessao.conteudo|sanitize|striptags }} |
|||
</p> |
|||
</fieldset> |
|||
{% endif %} |
|||
|
|||
@ -1,8 +1,9 @@ |
|||
{% load common_tags %} |
|||
{% if object.consideracoesfinais.conteudo %} |
|||
<fieldset> |
|||
<legend>Considerações Finais</legend> |
|||
<div style="border:0.5px solid #BAB4B1; border-radius: 10px; background-color: rgba(225, 225, 225, .8);"> |
|||
<p>{{object.consideracoesfinais.conteudo|safe}}</p> |
|||
<p>{{object.consideracoesfinais.conteudo|sanitize}}</p> |
|||
</div> |
|||
</fieldset> |
|||
<br /><br /><br /> |
|||
|
|||
@ -1,8 +1,9 @@ |
|||
{% load common_tags %} |
|||
{% if object.ocorrenciasessao.conteudo %} |
|||
<fieldset> |
|||
<legend>Ocorrências da Sessão</legend> |
|||
<div style="border:0.5px solid #BAB4B1; border-radius: 10px; background-color: rgba(225, 225, 225, .8);"> |
|||
<p>{{object.ocorrenciasessao.conteudo|safe}}</p> |
|||
<p>{{object.ocorrenciasessao.conteudo|sanitize}}</p> |
|||
</div> |
|||
</fieldset> |
|||
<br /><br /><br /> |
|||
|
|||
Loading…
Reference in new issue