RTF, parsed — not regex-stripped.
\uN? Unicode escapes are followed by an ANSI fallback character on purpose — strip backslashes with a regex and you duplicate every non-ASCII character in the document.
the-problem
RTF looks like it should be easy to strip — it's plain ASCII text with backslash control words — which is exactly why so many home-grown parsers get it wrong. Non-ASCII characters are hex-escaped as \'hh (a byte in the document's code page, set by an \ansicpg control word like \ansicpg1252 earlier in the file), or written as \uN? — a Unicode code point immediately followed by an ANSI fallback character meant for readers that don't support \u. A regex that just deletes backslash sequences leaves that fallback character behind, so every accented letter, curly quote, or em dash gets duplicated in the output. Embedded objects and images are stored as long hex blobs inline in the control-word stream, which a naive stripper will happily interpret as more 'text' if it isn't specifically recognized and skipped.
one-request-solution
txtfetch runs RTF through Apache Tika's actual RTF parser, which tracks the code page from \ansicpg, resolves \'hh hex escapes and \uN? Unicode-plus-fallback pairs correctly (keeping the Unicode character, dropping the fallback), and recognizes embedded-object hex blobs as binary data rather than text — so the response is clean prose, not a document with every special character doubled.
curl -X POST https://api.txtfetch.com/v1/extract \
-H "Authorization: Bearer $TXTFETCH_KEY" \
-F file=@signed-agreement.rtfimport os
import requests
with open("signed-agreement.rtf", "rb") as f:
r = requests.post(
"https://api.txtfetch.com/v1/extract",
headers={"Authorization": f"Bearer {os.environ['TXTFETCH_KEY']}"},
files={"file": f},
)
print(r.json()["extracted_text"])import { readFile } from "node:fs/promises";
const file = new Blob([await readFile("signed-agreement.rtf")]);
const form = new FormData();
form.append("file", file, "signed-agreement.rtf");
const res = await fetch("https://api.txtfetch.com/v1/extract", {
method: "POST",
headers: { Authorization: `Bearer ${process.env.TXTFETCH_KEY}` },
body: form,
});
const { extracted_text } = await res.json();
console.log(extracted_text);package main
import (
"bytes"
"encoding/json"
"fmt"
"io"
"mime/multipart"
"net/http"
"os"
)
type extractResponse struct {
Status string `json:"status"`
ExtractedText string `json:"extracted_text"`
}
func main() {
f, err := os.Open("signed-agreement.rtf")
if err != nil {
panic(err)
}
defer f.Close()
var body bytes.Buffer
writer := multipart.NewWriter(&body)
part, err := writer.CreateFormFile("file", "signed-agreement.rtf")
if err != nil {
panic(err)
}
if _, err := io.Copy(part, f); err != nil {
panic(err)
}
writer.Close()
req, err := http.NewRequest("POST", "https://api.txtfetch.com/v1/extract", &body)
if err != nil {
panic(err)
}
req.Header.Set("Authorization", "Bearer "+os.Getenv("TXTFETCH_KEY"))
req.Header.Set("Content-Type", writer.FormDataContentType())
resp, err := http.DefaultClient.Do(req)
if err != nil {
panic(err)
}
defer resp.Body.Close()
var result extractResponse
if err := json.NewDecoder(resp.Body).Decode(&result); err != nil {
panic(err)
}
fmt.Println(result.ExtractedText)
}Or skip the download — pass a url parameter and txtfetch fetches the document server-side:
curl -X POST "https://api.txtfetch.com/v1/extract?url=https://example.com/legal/terms-v3.rtf" \
-H "Authorization: Bearer $TXTFETCH_KEY"import os
import requests
r = requests.post(
"https://api.txtfetch.com/v1/extract",
headers={"Authorization": f"Bearer {os.environ['TXTFETCH_KEY']}"},
params={"url": "https://example.com/legal/terms-v3.rtf"},
)
print(r.json()["extracted_text"])const endpoint = new URL("https://api.txtfetch.com/v1/extract");
endpoint.searchParams.set("url", "https://example.com/legal/terms-v3.rtf");
const res = await fetch(endpoint, {
method: "POST",
headers: { Authorization: `Bearer ${process.env.TXTFETCH_KEY}` },
});
const { extracted_text } = await res.json();
console.log(extracted_text);package main
import (
"encoding/json"
"fmt"
"net/http"
"net/url"
"os"
)
type extractResponse struct {
Status string `json:"status"`
ExtractedText string `json:"extracted_text"`
}
func main() {
endpoint, err := url.Parse("https://api.txtfetch.com/v1/extract")
if err != nil {
panic(err)
}
q := endpoint.Query()
q.Set("url", "https://example.com/legal/terms-v3.rtf")
endpoint.RawQuery = q.Encode()
req, err := http.NewRequest("POST", endpoint.String(), nil)
if err != nil {
panic(err)
}
req.Header.Set("Authorization", "Bearer "+os.Getenv("TXTFETCH_KEY"))
resp, err := http.DefaultClient.Do(req)
if err != nil {
panic(err)
}
defer resp.Body.Close()
var result extractResponse
if err := json.NewDecoder(resp.Body).Decode(&result); err != nil {
panic(err)
}
fmt.Println(result.ExtractedText)
}{
"status": "success",
"extracted_text": "..."
}formats-covered
.rtf
faq
- Why do accented characters or curly quotes appear twice in my extracted RTF text?
- That's the classic symptom of naive backslash-stripping: RTF's \uN? Unicode escape is followed by a plain-ASCII fallback character by design, and a regex that just deletes backslash sequences leaves the fallback character behind next to the real one. A real RTF parser resolves the pair correctly instead.
- Does character encoding vary between RTF files?
- Yes — the code page for \'hh hex escapes is set per-document by a \ansicpg control word (commonly \ansicpg1252 for Windows-1252), so the same hex byte can mean a different character in different RTF files. Tika reads the declared code page rather than assuming one.
- What happens to embedded images or OLE objects in an RTF file?
- They're stored as hex-encoded binary blobs inline in the RTF stream. Tika's parser recognizes and skips them as binary data rather than attempting to read them as text.
go-further
Stop parsing. Start shipping.
Create an account and get an API key in minutes — the free Hobby plan needs no card.
Get started →