Determine whether text has a code-like structural profile.
The classifier computes a weighted score from several structural
signals:
- indentation whitespace as a percentage of total characters
- lines ending in a semicolon as a percentage of total lines
- symbol concentration
- code-like line prefixes as a percentage of total lines
- lines as a percentage of total characters
A Unix shebang adds a fixed bonus of 20 points.
The final score is compared against CODE_LIKE_SCORE_THRESHOLD to
determine whether the text is considered code-like.
Examples:
>>> is_code_like('printf("hello");\n')
True
>>> is_code_like("int main(void) {\n printf(\"hello\");\n return 0;\n}\n")
True
>>> is_code_like("The quick brown fox jumps over the lazy dog.\n")
False
>>> is_code_like(open("samples/sample_module.py").read())
True
>>> is_code_like(open("samples/sample_text.txt").read())
False
Parameters:
Returns:
-
bool
–
Whether the computed code-like score meets the threshold.
Source code in src/chunklet/self_tuning_chunker/utils.py
| def is_code_like(text: str) -> bool:
"""Determine whether text has a code-like structural profile.
The classifier computes a weighted score from several structural
signals:
- indentation whitespace as a percentage of total characters
- lines ending in a semicolon as a percentage of total lines
- symbol concentration
- code-like line prefixes as a percentage of total lines
- lines as a percentage of total characters
A Unix shebang adds a fixed bonus of 20 points.
The final score is compared against ``CODE_LIKE_SCORE_THRESHOLD`` to
determine whether the text is considered code-like.
Examples:
>>> is_code_like('printf("hello");\\n')
True
>>> is_code_like("int main(void) {\\n printf(\\"hello\\");\\n return 0;\\n}\\n")
True
>>> is_code_like("The quick brown fox jumps over the lazy dog.\\n")
False
>>> is_code_like(open("samples/sample_module.py").read())
True
>>> is_code_like(open("samples/sample_text.txt").read())
False
Args:
text: Text to classify.
Returns:
Whether the computed code-like score meets the threshold.
"""
if not text:
return False
lines = text.splitlines()
if not lines:
return False
line_count = len(lines)
weight = 0.0
if re.match(r"^#!/usr/bin/", text):
weight += 20
# Indentation density
indent_chars = sum(len(line) - len(line.lstrip(" \t")) for line in lines)
indent_percent = indent_chars / len(text) * 100
weight += indent_percent
# Symbol density
symbol_percent = (
sum(text.count(char) for char in CODE_SYMBOL_CHARS) / len(text) * 100
)
weight += symbol_percent
# Line-terminating semicolon density
semicolon_count = len(LINE_ENDING_SEMICOLON_PATTERN.findall(text))
semicolon_percent = semicolon_count / line_count * 100
weight += semicolon_percent
# Line density
newline_percent = line_count / len(text) * 100
weight += newline_percent * 2
# Code-like line-prefix density
prefix_count = len(CODE_LINE_PREFIX_PATTERN.findall(text))
prefix_percent = prefix_count / line_count * 100
weight += prefix_percent * 2
return weight >= CODE_LIKE_SCORE_THRESHOLD
|