"""Extrai a estrutura textual de um documento LaTeX (abntex2/ABNT). Converte o .tex em texto limpo preservando: - título, autores, local/data (elementos pré-textuais); - resumo e palavras-chave; - seções/subseções e seus parágrafos; - chaves de citação (\\cite, \\citeonline, \\citeauthor ...). Não depende de pandoc: usa apenas a biblioteca padrão. """ from __future__ import annotations import re from dataclasses import dataclass, field from pathlib import Path @dataclass class Section: """Seção do artigo com o texto limpo já convertido para prosa.""" title: str level: int = 1 @property def words(self) -> int: return len(count_words(self.text)) @dataclass class Article: title: str = "" authors: list[str] = field(default_factory=list) local: str = "" data: str = "" instituicao: str = "" abstract: str = "" keywords: list[str] = field(default_factory=list) sections: list[Section] = field(default_factory=list) citations: list[str] = field(default_factory=list) # ordem de aparição figures: list[str] = field(default_factory=list) # captions das figuras tables: list[str] = field(default_factory=list) # captions de quadros text_words: int = 0 def section(self, title: str) -> Section | None: for s in self.sections: if s.title.strip().lower() == title.strip().lower(): return s return None # --------------------------------------------------------------------------- # Conversão LaTeX -> texto # --------------------------------------------------------------------------- _BRACE_CMDS = ["textbf", "emph", "textit", "texttt", "underline", "bold", "itshape", "scshape", "MakeUppercase", "MakeLowercase"] _CITE_RE = re.compile(r"\\cite\w*\{([^}]*)\}") _BRACE_CMD_RE = { cmd: re.compile(r"\\" + cmd + r"\{([^{}]*)\}") for cmd in _BRACE_CMDS } # Chaves com barra escapada no LaTeX (ex.: "\_", "\#"). # "\," "\;" "\%" "\$" "\&" já são tratados pela regex de duas linhas acima. _ESCAPE_MAP = { "\\_": "_", "\\#": "#", "\\\\": "\n", } def count_words(text: str) -> list[str]: return re.findall( r"[A-Za-z0-9À-ÖØ-öø-ÿ]+(?:['’-][A-Za-z0-9À-ÖØ-öø-ÿ]+)*", text ) def _fold_brace_commands(s: str) -> str: changed = True while changed: changed = False for cmd, pat in _BRACE_CMD_RE.items(): s2, n = pat.subn(r"\1", s) if n: s, changed = s2, True return s def strip_latex(text: str, citations: list[str] | None = None) -> str: """Remove comandos e ambientes, preservando o conteúdo legível.""" # citações: captura as chaves e remove do texto def _cite(m: re.Match) -> str: if citations is not None: for key in m.group(1).split(","): key = key.strip() if key: citations.append(key) return " " text = _CITE_RE.sub(_cite, text) # rótulos/referências internas e figuras text = re.sub(r"\\label\{[^}]*\}", " ", text) text = re.sub(r"\\(ref|pageref|eqref)\{[^}]*\}", " (ref.) ", text) text = re.sub(r"\\includegraphics.*", " ", text) text = re.sub(r"\\noindent", " ", text) # comandos com argumento em chaves genéricos (legend, caption, textbf...) text = _fold_brace_commands(text) text = re.sub(r"\\(legend|caption|text|footnote)\{[^{}]*(?:\{[^{}]*\}[^{}]*)*\}", " ", text) # ambientes: manter marcadores de item, eliminar o resto text = re.sub(r"\\begin\{itemize\}", "\n- ", text) text = re.sub(r"\\begin\{enumerate\}", "\n1. ", text) text = re.sub(r"\\item\b", "\n- ", text) text = re.sub(r"\\(end|begin)\{[^}]*\}", " ", text) text = re.sub(r"\\toprule|\\midrule|\\bottomrule", " ", text) text = text.replace("\\\\", "\n") # quebras de linha (\\) em tabular/autor # comandos restantes sem argumento text = re.sub(r"\\[a-zA-Z]+\*?", " ", text) text = re.sub(r"\\[,;:\\!%$&]", " ", text) for esc, ch in _ESCAPE_MAP.items(): text = text.replace(esc, ch) # aspas e travessões text = text.replace("``", '"').replace("''", '"') text = text.replace("---", " — ").replace("--", " – ") # tabular: & vira coluna text = text.replace("&", " | ") # limpar pontuação e espaços text = re.sub(r"\[[a-z]*\]", " ", text) # posicionadores [htb] text = re.sub(r"[{}]+", " ", text) text = re.sub(r"[ \t]+", " ", text) text = re.sub(r" +([,.;:!?])", r"\1", text) text = re.sub(r"\n\s*", "\n", text) text = re.sub(r"\n{3,}", "\n\n", text) return text.strip() # --------------------------------------------------------------------------- # Parsing da estrutura # --------------------------------------------------------------------------- _SECTION_RE = re.compile(r"^\s*\\(section|subsection|subsubsection)\*?\s*\{([^}]+)\}") def _capture_command(text: str, name: str) -> str: """Extrai o último conteúdo de \\name{...} ou \\ ... \\end{}.""" m = re.search(r"\\" + name + r"\s*\{", text) if not m: return "" start = m.end() depth = 1 i = start while i < len(text) and depth: if text[i] == "{": depth += 1 elif text[i] == "}": depth -= 1 i += 1 return text[start : i - 1].strip() def _extract_authors(text: str) -> list[str]: raw = _capture_command(text, "autor") if not raw: raw = _capture_command(text, "author") return [a.strip() for a in re.split(r"\\\\|\\[ ]", raw) if a.strip()] def parse_article(tex_path: Path | str) -> Article: tex_path = Path(tex_path) text = tex_path.read_text(encoding="utf-8") # remove comentários de linha (não escapados) text = re.sub(r"(?