Spaces:
Running
Running
File size: 4,792 Bytes
fecd9f0 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 | import re
import gradio as gr
MAX_CHARS = 2_000
UTM_URL = (
"https://lynote.ai/ai-humanizer?utm_source=huggingface"
"&utm_medium=space&utm_campaign=hf_launch&utm_content=humanizer_lite"
)
EN_REPLACEMENTS = {
"in today's rapidly evolving world": "today",
"it is important to note that": "",
"it is worth noting that": "",
"serves as a testament to": "shows",
"delve into": "examine",
"leverage": "use",
"seamless": "smooth",
"robust": "reliable",
"game-changer": "useful change",
"in conclusion": "",
"moreover": "also",
"furthermore": "also",
}
ZH_REPLACEMENTS = {
"值得注意的是,": "",
"值得注意的是": "",
"综上所述,": "",
"综上所述": "",
"在当今快速发展的时代": "现在",
"赋能": "帮助",
"助力": "帮助",
"降本增效": "降低成本、提高效率",
"闭环": "完整流程",
"无缝": "顺畅",
}
PROTECTED_PATTERN = re.compile(
r"(`[^`]+`|https?://\S+|(?:[A-Za-z]:)?(?:[/\\][\w.\-]+)+|"
r"\b\d+(?:\.\d+)?(?:%|[A-Za-z]+)?\b|\"[^\"]*\"|'[^']*')"
)
def _protect(text: str):
protected = []
def replace(match):
token = f"⟦PROTECTED_{len(protected)}⟧"
protected.append(match.group(0))
return token
return PROTECTED_PATTERN.sub(replace, text), protected
def _restore(text: str, protected):
for index, value in enumerate(protected):
text = text.replace(f"⟦PROTECTED_{index}⟧", value)
return text
def _replace_case_insensitive(text: str, old: str, new: str):
return re.sub(re.escape(old), new, text, flags=re.IGNORECASE)
def humanize(text: str, voice: str):
original = (text or "").strip()
if not original:
return "Please enter some text.", "No text was processed."
if len(original) > MAX_CHARS:
return original, f"Input is limited to {MAX_CHARS:,} characters in this demo."
working, protected = _protect(original)
changes = []
for old, new in EN_REPLACEMENTS.items():
updated = _replace_case_insensitive(working, old, new)
if updated != working:
changes.append(f"Replaced formulaic phrase: {old}")
working = updated
for old, new in ZH_REPLACEMENTS.items():
if old in working:
working = working.replace(old, new)
changes.append(f"替换模板化表达:{old}")
working = re.sub(r"[ \t]{2,}", " ", working)
working = re.sub(r"\s+([,.;:!?,。;:!?])", r"\1", working)
working = re.sub(r"([.!?。!?])\s*\1+", r"\1", working)
working = re.sub(r"\n{3,}", "\n\n", working).strip(" ,,")
if voice == "Concise":
working = re.sub(r"\b(very|really|highly|extremely)\s+", "", working, flags=re.I)
working = working.replace("非常", "").replace("极其", "")
elif voice == "Professional":
working = re.sub(r"\b(a lot of)\b", "many", working, flags=re.I)
working = working.replace("挺多", "较多").replace("搞定", "完成")
result = _restore(working, protected)
if not changes and result == original:
note = "No high-confidence template phrases were changed."
else:
note = f"Applied {len(changes)} conservative style edit(s). Protected numbers, URLs, paths, code, and quotes."
return result, note
DESCRIPTION = f"""
This free, local demo removes a small set of high-confidence AI-writing clichés while
protecting numbers, URLs, paths, code, and quoted text. It is based on ideas from
[`humanize-text`](https://github.com/lynote-ai/humanize-text) and
[`humanize-text-skill`](https://github.com/lynote-ai/humanize-text-skill).
It is a writing-quality aid—not a guarantee that text will be classified as human.
[Try Lynote's full multi-stage humanizer]({UTM_URL}).
"""
with gr.Blocks(title="Lynote Humanizer Lite") as demo:
gr.Markdown("# ✍️ Lynote Humanizer Lite")
gr.Markdown(DESCRIPTION)
with gr.Row():
source = gr.Textbox(label="Draft", lines=14, max_lines=20, placeholder="Paste up to 2,000 characters…")
output = gr.Textbox(label="Edited draft", lines=14, max_lines=20)
voice = gr.Radio(["Natural", "Concise", "Professional"], value="Natural", label="Editing style")
run = gr.Button("Humanize conservatively", variant="primary")
notes = gr.Markdown()
run.click(humanize, inputs=[source, voice], outputs=[output, notes])
gr.Examples(
examples=[
["It is worth noting that this robust tool serves as a testament to our commitment to innovation.", "Natural"],
["值得注意的是,我们通过这套方案赋能团队,形成降本增效闭环。", "Concise"],
],
inputs=[source, voice],
)
if __name__ == "__main__":
demo.launch()
|