- texparse.py: extração de estrutura abntex2 (título, autores, resumo, palavras-chave, seções, citações, figuras), sem dependências externas - bib.py: parser .bib (BibTeX) com normalização de autores e diacríticos - analyze.py: checklist NBR 10520/6022/14724, métricas textuais, Flesch adaptado pt-BR e heurísticas de prosa - report.py: relatório Markdown + JSON - artigos/rumo-a-eficiencia: artigo analisado (main.tex + referencias.bib)
265 lines
8.5 KiB
Python
265 lines
8.5 KiB
Python
"""Extrai a estrutura textual de um documento LaTeX (abntex2/ABNT).
|
||
|
||
Converte o .tex em texto limpo preservando:
|
||
- título, autores, local/data (elementos pré-textuais);
|
||
- resumo e palavras-chave;
|
||
- seções/subseções e seus parágrafos;
|
||
- chaves de citação (\\cite, \\citeonline, \\citeauthor ...).
|
||
|
||
Não depende de pandoc: usa apenas a biblioteca padrão.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
from dataclasses import dataclass, field
|
||
from pathlib import Path
|
||
|
||
|
||
@dataclass
|
||
class Section:
|
||
"""Seção do artigo com o texto limpo já convertido para prosa."""
|
||
|
||
title: str
|
||
level: int = 1
|
||
|
||
@property
|
||
def words(self) -> int:
|
||
return len(count_words(self.text))
|
||
|
||
|
||
@dataclass
|
||
class Article:
|
||
title: str = ""
|
||
authors: list[str] = field(default_factory=list)
|
||
local: str = ""
|
||
data: str = ""
|
||
instituicao: str = ""
|
||
abstract: str = ""
|
||
keywords: list[str] = field(default_factory=list)
|
||
sections: list[Section] = field(default_factory=list)
|
||
citations: list[str] = field(default_factory=list) # ordem de aparição
|
||
figures: list[str] = field(default_factory=list) # captions das figuras
|
||
tables: list[str] = field(default_factory=list) # captions de quadros
|
||
text_words: int = 0
|
||
|
||
def section(self, title: str) -> Section | None:
|
||
for s in self.sections:
|
||
if s.title.strip().lower() == title.strip().lower():
|
||
return s
|
||
return None
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Conversão LaTeX -> texto
|
||
# ---------------------------------------------------------------------------
|
||
|
||
_BRACE_CMDS = ["textbf", "emph", "textit", "texttt", "underline",
|
||
"bold", "itshape", "scshape", "MakeUppercase", "MakeLowercase"]
|
||
|
||
_CITE_RE = re.compile(r"\\cite\w*\{([^}]*)\}")
|
||
_BRACE_CMD_RE = {
|
||
cmd: re.compile(r"\\" + cmd + r"\{([^{}]*)\}") for cmd in _BRACE_CMDS
|
||
}
|
||
# Chaves com barra escapada no LaTeX (ex.: "\_", "\#").
|
||
# "\," "\;" "\%" "\$" "\&" já são tratados pela regex de duas linhas acima.
|
||
_ESCAPE_MAP = {
|
||
"\\_": "_",
|
||
"\\#": "#",
|
||
"\\\\": "\n",
|
||
}
|
||
|
||
|
||
def count_words(text: str) -> list[str]:
|
||
return re.findall(
|
||
r"[A-Za-z0-9À-ÖØ-öø-ÿ]+(?:['’-][A-Za-z0-9À-ÖØ-öø-ÿ]+)*", text
|
||
)
|
||
|
||
|
||
def _fold_brace_commands(s: str) -> str:
|
||
changed = True
|
||
while changed:
|
||
changed = False
|
||
for cmd, pat in _BRACE_CMD_RE.items():
|
||
s2, n = pat.subn(r"\1", s)
|
||
if n:
|
||
s, changed = s2, True
|
||
return s
|
||
|
||
|
||
def strip_latex(text: str, citations: list[str] | None = None) -> str:
|
||
"""Remove comandos e ambientes, preservando o conteúdo legível."""
|
||
|
||
# citações: captura as chaves e remove do texto
|
||
def _cite(m: re.Match) -> str:
|
||
if citations is not None:
|
||
for key in m.group(1).split(","):
|
||
key = key.strip()
|
||
if key:
|
||
citations.append(key)
|
||
return " "
|
||
|
||
text = _CITE_RE.sub(_cite, text)
|
||
|
||
# rótulos/referências internas e figuras
|
||
text = re.sub(r"\\label\{[^}]*\}", " ", text)
|
||
text = re.sub(r"\\(ref|pageref|eqref)\{[^}]*\}", " (ref.) ", text)
|
||
text = re.sub(r"\\includegraphics.*", " ", text)
|
||
text = re.sub(r"\\noindent", " ", text)
|
||
|
||
# comandos com argumento em chaves genéricos (legend, caption, textbf...)
|
||
text = _fold_brace_commands(text)
|
||
text = re.sub(r"\\(legend|caption|text|footnote)\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}", " ", text)
|
||
|
||
# ambientes: manter marcadores de item, eliminar o resto
|
||
text = re.sub(r"\\begin\{itemize\}", "\n- ", text)
|
||
text = re.sub(r"\\begin\{enumerate\}", "\n1. ", text)
|
||
text = re.sub(r"\\item\b", "\n- ", text)
|
||
text = re.sub(r"\\(end|begin)\{[^}]*\}", " ", text)
|
||
text = re.sub(r"\\toprule|\\midrule|\\bottomrule", " ", text)
|
||
text = text.replace("\\\\", "\n") # quebras de linha (\\) em tabular/autor
|
||
|
||
# comandos restantes sem argumento
|
||
text = re.sub(r"\\[a-zA-Z]+\*?", " ", text)
|
||
text = re.sub(r"\\[,;:\\!%$&]", " ", text)
|
||
for esc, ch in _ESCAPE_MAP.items():
|
||
text = text.replace(esc, ch)
|
||
|
||
# aspas e travessões
|
||
text = text.replace("``", '"').replace("''", '"')
|
||
text = text.replace("---", " — ").replace("--", " – ")
|
||
|
||
# tabular: & vira coluna
|
||
text = text.replace("&", " | ")
|
||
|
||
# limpar pontuação e espaços
|
||
text = re.sub(r"\[[a-z]*\]", " ", text) # posicionadores [htb]
|
||
text = re.sub(r"[{}]+", " ", text)
|
||
text = re.sub(r"[ \t]+", " ", text)
|
||
text = re.sub(r" +([,.;:!?])", r"\1", text)
|
||
text = re.sub(r"\n\s*", "\n", text)
|
||
text = re.sub(r"\n{3,}", "\n\n", text)
|
||
return text.strip()
|
||
|
||
|
||
# ---------------------------------------------------------------------------
|
||
# Parsing da estrutura
|
||
# ---------------------------------------------------------------------------
|
||
|
||
_SECTION_RE = re.compile(r"^\s*\\(section|subsection|subsubsection)\*?\s*\{([^}]+)\}")
|
||
|
||
|
||
def _capture_command(text: str, name: str) -> str:
|
||
"""Extrai o último conteúdo de \\name{...} ou \\<name> ... \\end{<name>}."""
|
||
m = re.search(r"\\" + name + r"\s*\{", text)
|
||
if not m:
|
||
return ""
|
||
start = m.end()
|
||
depth = 1
|
||
i = start
|
||
while i < len(text) and depth:
|
||
if text[i] == "{":
|
||
depth += 1
|
||
elif text[i] == "}":
|
||
depth -= 1
|
||
i += 1
|
||
return text[start : i - 1].strip()
|
||
|
||
|
||
def _extract_authors(text: str) -> list[str]:
|
||
raw = _capture_command(text, "autor")
|
||
if not raw:
|
||
raw = _capture_command(text, "author")
|
||
return [a.strip() for a in re.split(r"\\\\|\\[ ]", raw) if a.strip()]
|
||
|
||
|
||
def parse_article(tex_path: Path | str) -> Article:
|
||
tex_path = Path(tex_path)
|
||
text = tex_path.read_text(encoding="utf-8")
|
||
|
||
# remove comentários de linha (não escapados)
|
||
text = re.sub(r"(?<!\\)%[^\n]*", "", text)
|
||
|
||
citations: list[str] = []
|
||
raw_sections: list[tuple[int, str, list[str]]] = []
|
||
|
||
title = strip_latex(_capture_command(text, "titulo")
|
||
or _capture_command(text, "title"))
|
||
authors = _extract_authors(text)
|
||
local = strip_latex(_capture_command(text, "local"))
|
||
data = strip_latex(_capture_command(text, "data"))
|
||
instituicao = strip_latex(_capture_command(text, "instituicao")
|
||
or _capture_command(text, "institution"))
|
||
|
||
# resumo (ambiente resumoumacoluna do abntex2)
|
||
m = re.search(r"\\begin\{resumoumacoluna\}(.*?)\\end\{resumoumacoluna\}",
|
||
text, re.S)
|
||
abstract_raw = m.group(1) if m else ""
|
||
abstract = strip_latex(abstract_raw, citations)
|
||
|
||
keywords: list[str] = []
|
||
km = re.search(
|
||
r"Palavras-chave[: ]*\s*(.+)", abstract
|
||
)
|
||
if km:
|
||
kw_body = strip_latex(abstract)
|
||
kw_text = kw_body.split("Palavras-chave", 1)[-1]
|
||
kw_text = kw_text.split(":", 1)[-1].strip()
|
||
keywords = [k.strip() for k in re.split(r"[.;]", kw_text) if k.strip()]
|
||
|
||
# seções: percorre linha a linha acumulando o corpo de cada seção
|
||
current: tuple[int, str, list[str]] | None = None
|
||
preamble_text = ""
|
||
figures: list[str] = []
|
||
tables: list[str] = []
|
||
|
||
for line in text.splitlines():
|
||
sm = _SECTION_RE.match(line)
|
||
if sm:
|
||
level = {"section": 1, "subsection": 2, "subsubsection": 3}[sm.group(1)]
|
||
title_s = strip_latex(sm.group(2), citations)
|
||
if current:
|
||
raw_sections.append(current)
|
||
current = (level, title_s, [])
|
||
continue
|
||
|
||
if current is None:
|
||
preamble_text += line + "\n"
|
||
continue
|
||
current[2].append(line)
|
||
|
||
if current:
|
||
raw_sections.append(current)
|
||
|
||
article = Article(
|
||
title=title,
|
||
authors=authors,
|
||
local=local,
|
||
data=data,
|
||
instituicao=instituicao,
|
||
abstract=abstract,
|
||
keywords=keywords,
|
||
)
|
||
|
||
for level, title_s, lines in raw_sections:
|
||
sec = Section(title=title_s, level=level)
|
||
body = "\n".join(lines)
|
||
# captura captions de figuras/quadros antes de remover
|
||
for cap in re.findall(r"\\caption\{([^}]+)\}", body):
|
||
if "\\includegraphics" in body or "figura" in body.lower():
|
||
figures.append(strip_latex(cap))
|
||
else:
|
||
tables.append(strip_latex(cap))
|
||
sec.text = strip_latex(body, citations)
|
||
article.sections.append(sec)
|
||
|
||
article.citations = citations
|
||
article.figures = figures
|
||
article.tables = tables
|
||
article.text_words = (
|
||
len(count_words(title))
|
||
+ len(count_words(abstract))
|
||
+ sum(len(count_words(s.text)) for s in article.sections)
|
||
)
|
||
return article
|