我终于编写了自己的辅助函数来获取原始字符串的跨度:
public HashMap<Integer, TokenSpan> getTokenSpans(String text, Tree parse)
{
List<String> tokens = new ArrayList<String>();
traverse(tokens, parse, parse.getChildrenAsList());
return extractTokenSpans(text, tokens);
}
private void traverse(List<String> tokens, Tree parse, List<Tree> children)
{
if(children == null)
return;
for(Tree child:children)
{
if(child.isLeaf())
{
tokens.add(child.value());
}
traverse(tokens, parse, child.getChildrenAsList());
}
}
private HashMap<Integer, TokenSpan> extractTokenSpans(String text, List<String> tokens)
{
HashMap<Integer, TokenSpan> result = new HashMap<Integer, TokenSpan>();
int spanStart, spanEnd;
int actCharIndex = 0;
int actTokenIndex = 0;
char actChar;
while(actCharIndex < text.length())
{
actChar = text.charAt(actCharIndex);
if(actChar == ' ')
{
actCharIndex++;
}
else
{
spanStart = actCharIndex;
String actToken = tokens.get(actTokenIndex);
int tokenCharIndex = 0;
while(tokenCharIndex < actToken.length() && text.charAt(actCharIndex) == actToken.charAt(tokenCharIndex))
{
tokenCharIndex++;
actCharIndex++;
}
if(tokenCharIndex != actToken.length())
{
//TODO: throw exception
}
actTokenIndex++;
spanEnd = actCharIndex;
result.put(actTokenIndex, new TokenSpan(spanStart, spanEnd));
}
}
return result;
}
然后我会打电话
getTokenSpans(originalString, parse)
所以我得到了一个映射,它可以将每个令牌映射到其对应的令牌跨度。这不是一个优雅的解决方案,但至少它有效。